diff --git a/R/explain.R b/R/explain.R index 091321d..5309e97 100644 --- a/R/explain.R +++ b/R/explain.R @@ -60,7 +60,15 @@ cog_explain <- function(result, format = c("print", "list")) { cli::cli_text("Basis: {prov$basis}{note}") } - if (!is.null(prov$expenditure_concept)) { + # Each verb reports its OWN concept. Both fields are always present (each + # defaults to its concept's default), so printing `expenditure_concept` + # unconditionally would tell a cog_revenue() caller "Concept: primary", + # which names a spending concept their result has nothing to do with. + if (identical(prov$verb, "cog_revenue")) { + if (!is.null(prov$revenue_concept)) { + cli::cli_text("Concept: {prov$revenue_concept} revenue") + } + } else if (!is.null(prov$expenditure_concept)) { concept_note <- if (!is.null(prov$expenditure_concept_note) && !is.na(prov$expenditure_concept_note)) { sprintf(" (%s)", prov$expenditure_concept_note) diff --git a/R/provenance.R b/R/provenance.R index 7b46220..77cf2d1 100644 --- a/R/provenance.R +++ b/R/provenance.R @@ -6,9 +6,10 @@ per_capita, adjust_to_year, result, sql, subtype_col, basis = NA_character_, basis_note = NA_character_, - expenditure_concept = "direct", + expenditure_concept = "primary", expenditure_concept_note = NA_character_, expenditure_concept_direct_suppressed = FALSE, + revenue_concept = "general", harmonization = NULL, recipe = NULL, suggestions = list(), completion = NULL) { @@ -67,6 +68,7 @@ expenditure_concept = expenditure_concept, expenditure_concept_note = expenditure_concept_note, expenditure_concept_direct_suppressed = isTRUE(expenditure_concept_direct_suppressed), + revenue_concept = revenue_concept, harmonization = harmonization %||% list( applied = FALSE, na_rows_excluded = 0L, na_amount_excluded = 0, note = NA_character_ diff --git a/R/revenue.R b/R/revenue.R index 49f7365..583752c 100644 --- a/R/revenue.R +++ b/R/revenue.R @@ -8,6 +8,31 @@ #' multiplies by 1000 and records the conversion in `provenance`). #' #' @inheritParams cog_spending +#' @param revenue_concept Which of Census's two published revenue concepts to +#' return. Concepts are defined as sets of the crosswalk's `revenue_subtype` +#' values -- never as item-code first letters, which cannot classify +#' correctly (prefix `Y` spans revenue, expenditure and balance codes, and +#' prefix `X` does the same): +#' +#' * `"general"` (default) -- Census General Revenue: `own_source` + +#' `federal` + `state` + `local_aid`. The manual defines this concept by +#' subtraction (section 4.3: *"General revenue comprises all revenue +#' except that classified as liquor store, utility, or insurance trust +#' revenue"*), so utility (`A91`-`A94`), liquor store (`A90`) and +#' insurance trust revenue are all excluded. +#' * `"total"` -- Census Total Revenue: every revenue subtype, i.e. +#' `general` plus utility, liquor store, and insurance trust revenue +#' (`Y01`/`Y02`/`Y04`/`Y11`/`Y12`/`Y51`/`Y52` and the employee-retirement +#' `X01`/`X02`/`X05`/`X08`). +#' +#' The two are related by Census's own identity, `Total Revenue = General + +#' Utility + Liquor Store + Insurance Trust`. +#' +#' Note that the employee-retirement (`X`) codes stop at FY2016, when those +#' systems moved out of the annual finance file into the separate Annual +#' Survey of Public Pensions, so a `"total"` series steps down at the +#' FY2016/FY2017 seam for reasons that are about collection scope rather +#' than revenue (series breaks `SB197`-`SB202`). #' @return Tibble with columns `year`, `canonical_govid`, `gov_name`, #' `revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`, #' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`, @@ -17,6 +42,7 @@ cog_revenue <- function(govid, years, category = NULL, per_capita = FALSE, adjust_to_year = NULL, basis = c("harmonized", "raw"), recipe = NULL, + revenue_concept = c("general", "total"), complete = FALSE) { # flow_prefixes no longer classifies rows (crosswalk revenue_subtype # membership does -- General Revenue, i.e. everything except @@ -35,6 +61,7 @@ cog_revenue <- function(govid, years, category = NULL, adjust_to_year = adjust_to_year, basis = basis, recipe = recipe, + revenue_concept = revenue_concept, complete = complete ) } diff --git a/R/spending.R b/R/spending.R index c56196f..9642d42 100644 --- a/R/spending.R +++ b/R/spending.R @@ -31,9 +31,26 @@ ) } -# cog_revenue()'s single concept (until uscogdata#12 adds more): Census -# General Revenue -- every crosswalk revenue subtype except insurance_trust. +# The two revenue concepts (uscogdata#12), again as crosswalk subtype sets. +# Census's manual section 4.3 defines the first by SUBTRACTING from the second +# -- "General revenue comprises all revenue except that classified as liquor +# store, utility, or insurance trust revenue" -- giving the identity +# +# Total Revenue = General + Utility + Liquor Store + Insurance Trust +# +# Verified against Census's own computed concept fields (IndFin FY2012, +# Wisconsin state): 31,410,686 + 0 + 0 + 4,469,906 = 35,880,592, exact. .revenue_subtypes_general <- c("own_source", "federal", "state", "local_aid") +.revenue_subtypes_total <- c(.revenue_subtypes_general, "utility", + "liquor_store", "insurance_trust") + +#' @noRd +.revenue_concept_subtypes <- function(concept) { + switch(concept, + general = .revenue_subtypes_general, + total = .revenue_subtypes_total + ) +} #' Summarized spending by category #' @@ -198,6 +215,7 @@ cog_spending <- function(govid, years, category = NULL, per_capita, adjust_to_year, basis = c("harmonized", "raw"), recipe = NULL, expenditure_concept = c("primary", "direct", "total"), + revenue_concept = c("general", "total"), complete = FALSE) { basis_explicit <- length(basis) == 1L basis <- match.arg(basis, c("harmonized", "raw")) @@ -215,16 +233,27 @@ cog_spending <- function(govid, years, category = NULL, } ) + revenue_concept <- tryCatch( + match.arg(revenue_concept, c("general", "total")), + error = function(e) { + cli::cli_abort( + "`revenue_concept` must be one of {.val general} or {.val total}.", + class = "uscogdata_invalid_revenue_concept", + parent = e + ) + } + ) + # The concept's subtype scope. Every code path below -- the verb SQL, the # harmonization exclusion count, and the complete = TRUE grid -- is scoped - # by crosswalk subtype membership, never by item-code prefix. For revenue - # there is a single concept today (General Revenue; uscogdata#12 will add - # more). "total"'s extra intergovernmental leg travels through the ig_* - # views, not through this scope. + # by crosswalk subtype membership, never by item-code prefix. The + # expenditure "total" concept's extra intergovernmental leg is the one + # exception: it travels through the ig_* views rather than this scope, + # because its legacy rows are aggregate-flagged. subtype_scope <- if (identical(subtype_col, "spend_subtype")) { .expenditure_concept_subtypes(expenditure_concept) } else { - .revenue_subtypes_general + .revenue_concept_subtypes(revenue_concept) } govid <- .coerce_govid_input(govid, arg = "govid") @@ -427,6 +456,7 @@ cog_spending <- function(govid, years, category = NULL, expenditure_concept = expenditure_concept, expenditure_concept_note = expenditure_concept_note_for_prov, expenditure_concept_direct_suppressed = direct_suppressed_flag, + revenue_concept = revenue_concept, harmonization = harmonization, recipe = recipe_block, suggestions = suggestions, diff --git a/README.md b/README.md index 8d2f2d9..16cbdd5 100644 --- a/README.md +++ b/README.md @@ -71,6 +71,38 @@ enforce this by refusing `expenditure_concept = "total"`. See `vignette("total-spending", package = "uscogdata")` for the full explanation with worked examples. +## General vs Total revenue + +`cog_revenue(..., revenue_concept = c("general", "total"))` selects between +Census's two published revenue concepts, again defined as crosswalk +`revenue_subtype` sets rather than item-code prefixes: + +- `"general"` (the default) is Census **General Revenue**: own-source + (taxes, charges, miscellaneous) plus federal, state and local + intergovernmental aid. +- `"total"` is Census **Total Revenue**: `general` plus utility revenue + (`A91`–`A94`), liquor store revenue (`A90`), and insurance trust revenue + (unemployment and workers' compensation `Y` codes plus the + employee-retirement `X` codes). + +The manual defines the first by subtracting the other three from the second, +so the two are related by Census's own identity: + +``` +Total Revenue = General + Utility + Liquor Store + Insurance Trust +``` + +Two things worth knowing before switching to `"total"`: + +- **Utility revenue is large for cities.** Measured on the bundled fixture, + utility plus liquor store revenue is 15.9% of city (type 2) revenue, versus + 1.2% for states and 1.7% for counties. `general` excludes it by definition. +- **The employee-retirement (`X`) codes stop at FY2016**, when those systems + moved out of the annual finance file into the separate Annual Survey of + Public Pensions. A `"total"` series therefore steps down at the + FY2016/FY2017 seam for reasons of collection scope, not revenue (series + breaks `SB197`–`SB202`, in the corpus's `series_breaks` table). + ## Developer notes ### Testing diff --git a/inst/extdata/fixture_corpus/data/series_breaks.parquet b/inst/extdata/fixture_corpus/data/series_breaks.parquet index eba2c94..0351bbd 100644 Binary files a/inst/extdata/fixture_corpus/data/series_breaks.parquet and b/inst/extdata/fixture_corpus/data/series_breaks.parquet differ diff --git a/inst/extdata/fixture_corpus/data/summary_categories.parquet b/inst/extdata/fixture_corpus/data/summary_categories.parquet index 130944e..87a965a 100644 Binary files a/inst/extdata/fixture_corpus/data/summary_categories.parquet and b/inst/extdata/fixture_corpus/data/summary_categories.parquet differ diff --git a/inst/extdata/fixture_corpus/manifest.json b/inst/extdata/fixture_corpus/manifest.json index a8cfd11..baa0ef5 100644 --- a/inst/extdata/fixture_corpus/manifest.json +++ b/inst/extdata/fixture_corpus/manifest.json @@ -1,7 +1,7 @@ { "schema_version": 6, - "built_at": "2026-07-30T20:01:56Z", - "pipeline_commit": "e64a046", + "built_at": "2026-07-31T00:47:27Z", + "pipeline_commit": "aadb46b", "fixture_note": "Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated from the sparsified schema-v6 corpus: the wide era (<= FY2011) no longer stores explicit zeros, so FY2011 absence means Census published $0 while FY2012+ absence means not reported. representation.parquet and code_set.parquet carry that rule and ship in full, as do every other metadata table in the publish tree. 2011/2012 straddle both the wide-aggregate -> modern-leaf format boundary (exercised by basis=\"harmonized\" and recipe= queries) and the dense -> sparse representation boundary (SB194); 2019/2020 retain the prior per-capita/CPI regression anchors. Regenerated via data-raw/regenerate_fixture_corpus.R.", "data_vintage": { "source_vintages": { @@ -105,12 +105,12 @@ }, { "path": "data/series_breaks.parquet", - "sha256": "381090660c8b8a1bee852e7f63d29b9ecaf10f71870f41c63de92017c83b6f1f", + "sha256": "06dcc995ff533e57cc65fa25086cc9bf83ba592c58bf7cc99269dc2576f69944", "description": "series_breaks.parquet" }, { "path": "data/summary_categories.parquet", - "sha256": "e4918abf8e9e6d1372d7ddc255dc199f9c303e250f4106449241b48c68abee67", + "sha256": "e3b0efa00ce713b8f45829b89cfde24b55333f26101f0495df82d85997d18d8e", "description": "summary_categories.parquet" } ] diff --git a/inst/schemas/provenance-v1.json b/inst/schemas/provenance-v1.json index 9f3973b..779fad6 100644 --- a/inst/schemas/provenance-v1.json +++ b/inst/schemas/provenance-v1.json @@ -25,6 +25,12 @@ "type": "boolean", "description": "TRUE when expenditure_concept = 'total' and at least one requested (year, category) has intergovernmental rows but NO Direct rows in this corpus (typically a legacy aggregate-only family) -- those result rows report the intergovernmental leg alone, not Direct + IG. Always FALSE for expenditure_concept = 'primary' or 'direct'. See the affected rows' `notes` for the recovering recipe, if any." }, + "revenue_concept": { + "type": "string", + "enum": ["general", "total"], + "description": "Which revenue concept produced this result, defined as crosswalk revenue_subtype sets (never item-code prefixes). 'general' (the default) is Census General Revenue: own_source + federal + state + local_aid. 'total' is Census Total Revenue: general plus utility, liquor store and insurance trust revenue. Census defines the first by subtracting the other three from the second (manual section 4.3). Meaningful for cog_revenue() results; spending results carry the default.", + "$comment": "The employee-retirement (X) codes inside insurance_trust stop at FY2016, so a 'total' series steps at the FY2016/FY2017 seam for collection-scope reasons (series breaks SB197-SB202)." + }, "harmonization": { "type": "object" }, "recipe": { "type": ["object", "null"] }, "suggestions": { "type": "array" }, diff --git a/inst/sql/21-revenue_long.sql b/inst/sql/21-revenue_long.sql index 61e6c88..e884a05 100644 --- a/inst/sql/21-revenue_long.sql +++ b/inst/sql/21-revenue_long.sql @@ -1,15 +1,18 @@ -- Revenue rows, classified by crosswalk MEMBERSHIP rather than item-code -- first letter (see 20-spending_long.sql for why prefixes cannot work). --- Scope is Census General Revenue: every crosswalk revenue subtype EXCEPT --- insurance_trust (Y01/Y02/Y04/Y11/Y12/Y51/Y52). Owner ruling 2026-07-30: --- the default revenue concept stays general; surfacing insurance-trust --- revenue through an explicit concept argument is uscogdata#12. +-- +-- Carries EVERY revenue subtype. Which of Census's two published concepts a +-- query actually returns is decided per revenue_concept in R +-- (.verb_spendrev), exactly as expenditure_concept narrows spending_long: +-- general = own_source + federal + state + local_aid (the default) +-- total = general + utility + liquor_store + insurance_trust +-- Census defines the first by subtracting the other three from the second +-- (manual section 4.3), so both concepts need all four families present here. CREATE OR REPLACE VIEW revenue_long AS SELECT * FROM long WHERE item_code IN ( SELECT item_code FROM summary_categories WHERE category_type = 'revenue' - AND revenue_subtype <> 'insurance_trust' ) AND NOT is_aggregate; diff --git a/inst/sql/23-revenue_long_harmonized.sql b/inst/sql/23-revenue_long_harmonized.sql index 3a5492c..4f61a5c 100644 --- a/inst/sql/23-revenue_long_harmonized.sql +++ b/inst/sql/23-revenue_long_harmonized.sql @@ -1,5 +1,5 @@ -- Harmonized-basis twin of 21-revenue_long.sql: same crosswalk-membership --- classification (General Revenue = revenue minus insurance_trust), applied +-- classification (every revenue subtype; the concept narrows in R), applied -- to harmonized_code rather than the published item_code. CREATE OR REPLACE VIEW revenue_long_harmonized AS SELECT * REPLACE (harmonized_code AS item_code) @@ -9,5 +9,4 @@ WHERE NOT is_aggregate AND harmonized_code IN ( SELECT item_code FROM summary_categories WHERE category_type = 'revenue' - AND revenue_subtype <> 'insurance_trust' ); diff --git a/man/cog_revenue.Rd b/man/cog_revenue.Rd index b018fef..753dede 100644 --- a/man/cog_revenue.Rd +++ b/man/cog_revenue.Rd @@ -12,6 +12,7 @@ cog_revenue( adjust_to_year = NULL, basis = c("harmonized", "raw"), recipe = NULL, + revenue_concept = c("general", "total"), complete = FALSE ) } @@ -57,6 +58,32 @@ argument is ignored and the result's provenance reports FALSE`, pointing at the `recipe` block instead) rather than a possibly-misleading `"harmonized"`/`"raw"` value.} +\item{revenue_concept}{Which of Census's two published revenue concepts to + return. Concepts are defined as sets of the crosswalk's `revenue_subtype` + values -- never as item-code first letters, which cannot classify + correctly (prefix `Y` spans revenue, expenditure and balance codes, and + prefix `X` does the same): + + * `"general"` (default) -- Census General Revenue: `own_source` + + `federal` + `state` + `local_aid`. The manual defines this concept by + subtraction (section 4.3: *"General revenue comprises all revenue + except that classified as liquor store, utility, or insurance trust + revenue"*), so utility (`A91`-`A94`), liquor store (`A90`) and + insurance trust revenue are all excluded. + * `"total"` -- Census Total Revenue: every revenue subtype, i.e. + `general` plus utility, liquor store, and insurance trust revenue + (`Y01`/`Y02`/`Y04`/`Y11`/`Y12`/`Y51`/`Y52` and the employee-retirement + `X01`/`X02`/`X05`/`X08`). + + The two are related by Census's own identity, `Total Revenue = General + + Utility + Liquor Store + Insurance Trust`. + + Note that the employee-retirement (`X`) codes stop at FY2016, when those + systems moved out of the annual finance file into the separate Annual + Survey of Public Pensions, so a `"total"` series steps down at the + FY2016/FY2017 seam for reasons that are about collection scope rather + than revenue (series breaks `SB197`-`SB202`).} + \item{complete}{If `TRUE`, fill the requested grid so that a cell the corpus does not carry still appears, labelled with **why** it is missing, and add a `value_source` column to every row: diff --git a/tests/testthat/test-categories.R b/tests/testthat/test-categories.R index 4f89219..bed3100 100644 --- a/tests/testthat/test-categories.R +++ b/tests/testthat/test-categories.R @@ -49,12 +49,14 @@ test_that("cog_categories(type = 'revenue') returns only revenue rows", { skip_if_no_corpus() r <- cog_categories(type = "revenue") expect_true(all(r$category_type == "revenue")) - # `insurance_trust` (Y01/Y02/Y04/Y11/Y12/Y51/Y52) is deliberately NOT - # own_source: Census's "General Revenue" excludes insurance trust revenue, - # and Y01 alone is $1.31T corpus-wide. + # The four non-general subtypes are deliberately NOT own_source: Census's + # General Revenue excludes insurance trust (Y01 alone is $1.31T corpus-wide, + # plus the employee-retirement X codes), utility (A91-A94) and liquor store + # (A90) revenue by definition, which is what makes both of its published + # revenue concepts computable -- see `revenue_concept` in `?cog_revenue`. expect_true(all(r$subtype %in% c("own_source", "federal", "state", "local_aid", - "insurance_trust"))) + "insurance_trust", "utility", "liquor_store"))) }) test_that("cog_categories(pattern = ...) filters case-insensitively", { diff --git a/tests/testthat/test-revenue-concept-insurance-trust.R b/tests/testthat/test-revenue-concept-insurance-trust.R index e53d9d2..068d0f1 100644 --- a/tests/testthat/test-revenue-concept-insurance-trust.R +++ b/tests/testthat/test-revenue-concept-insurance-trust.R @@ -9,47 +9,131 @@ # a published Census revenue concept exactly the way I89 sits inside Census's # Direct Expenditure concept (finding F-012). # -# CAVEAT FOR WHOEVER PICKS THIS UP: the argument name below (`revenue_concept = -# "total"`) is this test's *proposal*, not a settled decision. The owner's -# 2026-07-28 resolution covers expenditure concepts only; no revenue-side -# naming has been ruled on. If the eventual argument is named differently, -# change the two calls here -- the asserted dollar invariants are what matter -# and are independent of the naming. +# RULED 2026-07-30. `revenue_concept = c("general", "total")` mirrors +# `expenditure_concept`, and the two values are Census's two published revenue +# concepts, related by the manual's own identity (section 4.3, which defines +# the first by SUBTRACTING from the second): +# +# Total Revenue = General + Utility + Liquor Store + Insurance Trust +# +# so `general` is the four general subtypes (own_source/federal/state/ +# local_aid) and `total` is every revenue subtype. Naming utility (A91-A94) +# and liquor store (A90) separately is what makes BOTH computable -- before +# cog_pipeline#79 they sat in own_source, so the default was really +# "General + Utility + Liquor", a concept Census does not publish. # # Fixture reproducibility: Madison's own X-prefix revenue (FY1970-FY1986, # $15,098,000 nominal, $0 thereafter) is outside the bundled fixture's year # window (2011/2012/2019/2020), so the same invariant is asserted on Wisconsin -# state government FY2012, where the fixture carries nonzero X01/X05/X08. +# state government FY2012, where the fixture carries nonzero X01/X02/X05/X08. test_that("cog_revenue() can return Census Total Revenue including Insurance Trust (prefix X)", { - testthat::skip("Blocked on uscogdata#12 (finding F-014)") - wi_state <- "550000227544" # WISCONSIN (state government) # Revenue-shaped Employee Retirement codes, read from the RAW corpus rather # than through cog_revenue(), which is the filter under test: - # X01 local employee contribution, X04/X05 contributions and transfers from - # other governments, X08 earnings on investments. - x_revenue <- wt_raw_amt(wi_state, 2012L, codes = c("X01", "X04", "X05", "X08")) - expect_equal(x_revenue, 2038800) # 615,835 + 0 + 560,382 + 862,583 ($1,000s) + # X01/X02 employee contributions, X05 contributions from other governments, + # X08 total earnings on investments. + # + # X04 and X06 are deliberately NOT in this set, though an earlier draft of + # this test included X04. Both are exhibit codes for INTRAgovernmental + # transfers (the administering government paying into its own fund), which + # X05's own definition excludes by name. Census agrees: its computed "Total + # Emp Ret Rev" for this government-year is exactly the four codes below. + x_revenue <- wt_raw_amt(wi_state, 2012L, codes = c("X01", "X02", "X05", "X08")) + expect_equal(x_revenue, 2283883) # 615,835 + 245,083 + 560,382 + 862,583 + + # The Y-prefix insurance trust revenue (unemployment + workers comp), which + # is the other half of the same Census concept. + y_revenue <- wt_raw_amt(wi_state, 2012L, codes = c("Y01", "Y11")) + expect_equal(y_revenue, 1259785) general <- cog_revenue(govid = wi_state, years = 2012L) + expect_equal(attr(general, "provenance")$revenue_concept, "general") expect_equal(sum(general$amt_nominal), 31338293000) total <- cog_revenue(govid = wi_state, years = 2012L, revenue_concept = "total") - expect_equal(sum(total$amt_nominal) - sum(general$amt_nominal), x_revenue * 1000) - expect_equal(sum(total$amt_nominal), 33377093000) - expect_true(all(c("X01", "X05", "X08") %in% wt_codes_included(total))) + expect_equal(attr(total, "provenance")$revenue_concept, "total") + + # total - general is the whole insurance trust leg, X and Y together. + # Asserted as a delta as well as a level so this stays correct however the + # utility/liquor families land (both are $0 for WI state in FY2012). + expect_equal(sum(total$amt_nominal) - sum(general$amt_nominal), + (x_revenue + y_revenue) * 1000) + expect_equal(sum(total$amt_nominal), 34881961000) + expect_true(all(c("X01", "X02", "X05", "X08") %in% wt_codes_included(total))) # Sibling codes under the SAME first letter must stay out: X11/X12 are # benefit payments (an expenditure) and X21/X30/X47 are cash and securities # holdings (a balance-sheet stock). This is the F-018 point restated on the - # revenue side -- the split has to come from the crosswalk's spend_type, not - # from the letter X. + # revenue side -- the split comes from the crosswalk, not from the letter X. expect_false(any(c("X11", "X12", "X21", "X30", "X47") %in% wt_codes_included(total))) - # Every returned row still resolves to a category. summary_categories has - # zero rows for prefix X today, so relaxing the prefix filter alone would - # produce category = NA rows -- see census_of_governments_finance_pipeline#60. + # Every returned row still resolves to a category (cog_pipeline#79 added the + # X crosswalk rows; relaxing a prefix filter alone would have produced + # category = NA rows). expect_false(any(is.na(total$category))) }) + +test_that("revenue_concept = 'general' is the default and is strict Census General Revenue", { + wi_state <- "550000227544" + default <- cog_revenue(govid = wi_state, years = 2012L) + explicit <- cog_revenue(govid = wi_state, years = 2012L, + revenue_concept = "general") + expect_equal(sum(default$amt_nominal), sum(explicit$amt_nominal)) + + # General Revenue excludes utility, liquor store AND insurance trust + # revenue. WI state carries $0 of utility/liquor in FY2012, so the level + # assertion above cannot see those two -- assert the subtype scope directly. + # + # A subset, not setequal: `state` means "intergovernmental revenue FROM the + # state government" (the C codes), which a STATE government does not receive + # from itself, so it is legitimately absent here. + expect_true(all(default$revenue_subtype %in% + c("own_source", "federal", "state", "local_aid"))) + expect_false(any(c("utility", "liquor_store", "insurance_trust") %in% + default$revenue_subtype)) +}) + +test_that("utility and liquor store revenue are inside `total` and outside `general`", { + # A city, where utility revenue is material: this is the case the WI state + # baseline structurally cannot exercise. Measured on the fixture, utility + + # liquor is 15.9% of what cog_revenue() returned for type-2 governments + # before the general/total split, so this is the largest behaviour change + # the concept split introduces. + con <- uscogdata:::.ensure_session() + gov <- DBI::dbGetQuery(con, + "SELECT canonical_govid, SUM(amt) amt FROM long + WHERE year = 2012 AND type = 2 AND NOT is_aggregate + AND item_code IN ('A91','A92','A93','A94') + GROUP BY 1 ORDER BY amt DESC LIMIT 1")$canonical_govid + + util_raw <- wt_raw_amt(gov, 2012L, codes = c("A90", "A91", "A92", "A93", "A94")) + expect_gt(util_raw, 0) + + general <- cog_revenue(govid = gov, years = 2012L) + total <- cog_revenue(govid = gov, years = 2012L, revenue_concept = "total") + + expect_false(any(c("utility", "liquor_store") %in% general$revenue_subtype)) + expect_true("utility" %in% total$revenue_subtype) + expect_equal(sum(total$amt_nominal) - sum(general$amt_nominal), + util_raw * 1000 + + wt_raw_amt(gov, 2012L, codes = c("Y01", "Y11", "X01", "X02", + "X05", "X08")) * 1000) +}) + +test_that("revenue_concept rejects unknown values and never returns a balance row", { + expect_error( + cog_revenue("550000227544", years = 2012L, revenue_concept = "gross"), + class = "uscogdata_invalid_revenue_concept" + ) + + # uscogdata#25 restated for the widest revenue concept: stocks are not + # flows, and `total` must not quietly admit the X/Y/W/Z balance families. + con <- uscogdata:::.ensure_session() + balance <- DBI::dbGetQuery(con, + "SELECT item_code, category FROM summary_categories WHERE category_type = 'balance'") + total <- cog_revenue("550000227544", years = 2012L, revenue_concept = "total") + expect_false(any(total$category %in% balance$category)) + expect_length(intersect(wt_codes_included(total), balance$item_code), 0L) +}) diff --git a/tests/testthat/test-views.R b/tests/testthat/test-views.R index 1153247..654c744 100644 --- a/tests/testthat/test-views.R +++ b/tests/testthat/test-views.R @@ -393,19 +393,19 @@ test_that("spending_long carries exactly the non-IG expenditure crosswalk codes expect_equal(agg_count, 0) }) -test_that("revenue_long carries exactly the general-revenue crosswalk codes and excludes aggregates", { +test_that("revenue_long carries exactly the revenue crosswalk codes and excludes aggregates", { skip_if_no_corpus() con <- cog_open() on.exit(cog_close()) - # General Revenue scope: revenue crosswalk members minus insurance_trust - # (owner ruling 2026-07-30; an explicit wider concept is uscogdata#12). + # The view carries EVERY revenue subtype; which of Census's two published + # concepts a query returns is decided per `revenue_concept` in R + # (uscogdata#12), exactly as `expenditure_concept` narrows spending_long. stray <- DBI::dbGetQuery(con, "SELECT DISTINCT s.item_code FROM revenue_long s LEFT JOIN summary_categories c USING (item_code) - WHERE c.category_type IS DISTINCT FROM 'revenue' - OR c.revenue_subtype = 'insurance_trust'" + WHERE c.category_type IS DISTINCT FROM 'revenue'" )$item_code expect_length(stray, 0L)