feat: revenue_concept = c("general", "total") off the crosswalk (#12) #27
+9
-1
@@ -60,7 +60,15 @@ cog_explain <- function(result, format = c("print", "list")) {
|
|||||||
cli::cli_text("Basis: {prov$basis}{note}")
|
cli::cli_text("Basis: {prov$basis}{note}")
|
||||||
}
|
}
|
||||||
|
|
||||||
if (!is.null(prov$expenditure_concept)) {
|
# Each verb reports its OWN concept. Both fields are always present (each
|
||||||
|
# defaults to its concept's default), so printing `expenditure_concept`
|
||||||
|
# unconditionally would tell a cog_revenue() caller "Concept: primary",
|
||||||
|
# which names a spending concept their result has nothing to do with.
|
||||||
|
if (identical(prov$verb, "cog_revenue")) {
|
||||||
|
if (!is.null(prov$revenue_concept)) {
|
||||||
|
cli::cli_text("Concept: {prov$revenue_concept} revenue")
|
||||||
|
}
|
||||||
|
} else if (!is.null(prov$expenditure_concept)) {
|
||||||
concept_note <- if (!is.null(prov$expenditure_concept_note) &&
|
concept_note <- if (!is.null(prov$expenditure_concept_note) &&
|
||||||
!is.na(prov$expenditure_concept_note)) {
|
!is.na(prov$expenditure_concept_note)) {
|
||||||
sprintf(" (%s)", prov$expenditure_concept_note)
|
sprintf(" (%s)", prov$expenditure_concept_note)
|
||||||
|
|||||||
+3
-1
@@ -6,9 +6,10 @@
|
|||||||
per_capita, adjust_to_year, result, sql,
|
per_capita, adjust_to_year, result, sql,
|
||||||
subtype_col, basis = NA_character_,
|
subtype_col, basis = NA_character_,
|
||||||
basis_note = NA_character_,
|
basis_note = NA_character_,
|
||||||
expenditure_concept = "direct",
|
expenditure_concept = "primary",
|
||||||
expenditure_concept_note = NA_character_,
|
expenditure_concept_note = NA_character_,
|
||||||
expenditure_concept_direct_suppressed = FALSE,
|
expenditure_concept_direct_suppressed = FALSE,
|
||||||
|
revenue_concept = "general",
|
||||||
harmonization = NULL, recipe = NULL,
|
harmonization = NULL, recipe = NULL,
|
||||||
suggestions = list(),
|
suggestions = list(),
|
||||||
completion = NULL) {
|
completion = NULL) {
|
||||||
@@ -67,6 +68,7 @@
|
|||||||
expenditure_concept = expenditure_concept,
|
expenditure_concept = expenditure_concept,
|
||||||
expenditure_concept_note = expenditure_concept_note,
|
expenditure_concept_note = expenditure_concept_note,
|
||||||
expenditure_concept_direct_suppressed = isTRUE(expenditure_concept_direct_suppressed),
|
expenditure_concept_direct_suppressed = isTRUE(expenditure_concept_direct_suppressed),
|
||||||
|
revenue_concept = revenue_concept,
|
||||||
harmonization = harmonization %||% list(
|
harmonization = harmonization %||% list(
|
||||||
applied = FALSE, na_rows_excluded = 0L, na_amount_excluded = 0,
|
applied = FALSE, na_rows_excluded = 0L, na_amount_excluded = 0,
|
||||||
note = NA_character_
|
note = NA_character_
|
||||||
|
|||||||
+27
@@ -8,6 +8,31 @@
|
|||||||
#' multiplies by 1000 and records the conversion in `provenance`).
|
#' multiplies by 1000 and records the conversion in `provenance`).
|
||||||
#'
|
#'
|
||||||
#' @inheritParams cog_spending
|
#' @inheritParams cog_spending
|
||||||
|
#' @param revenue_concept Which of Census's two published revenue concepts to
|
||||||
|
#' return. Concepts are defined as sets of the crosswalk's `revenue_subtype`
|
||||||
|
#' values -- never as item-code first letters, which cannot classify
|
||||||
|
#' correctly (prefix `Y` spans revenue, expenditure and balance codes, and
|
||||||
|
#' prefix `X` does the same):
|
||||||
|
#'
|
||||||
|
#' * `"general"` (default) -- Census General Revenue: `own_source` +
|
||||||
|
#' `federal` + `state` + `local_aid`. The manual defines this concept by
|
||||||
|
#' subtraction (section 4.3: *"General revenue comprises all revenue
|
||||||
|
#' except that classified as liquor store, utility, or insurance trust
|
||||||
|
#' revenue"*), so utility (`A91`-`A94`), liquor store (`A90`) and
|
||||||
|
#' insurance trust revenue are all excluded.
|
||||||
|
#' * `"total"` -- Census Total Revenue: every revenue subtype, i.e.
|
||||||
|
#' `general` plus utility, liquor store, and insurance trust revenue
|
||||||
|
#' (`Y01`/`Y02`/`Y04`/`Y11`/`Y12`/`Y51`/`Y52` and the employee-retirement
|
||||||
|
#' `X01`/`X02`/`X05`/`X08`).
|
||||||
|
#'
|
||||||
|
#' The two are related by Census's own identity, `Total Revenue = General +
|
||||||
|
#' Utility + Liquor Store + Insurance Trust`.
|
||||||
|
#'
|
||||||
|
#' Note that the employee-retirement (`X`) codes stop at FY2016, when those
|
||||||
|
#' systems moved out of the annual finance file into the separate Annual
|
||||||
|
#' Survey of Public Pensions, so a `"total"` series steps down at the
|
||||||
|
#' FY2016/FY2017 seam for reasons that are about collection scope rather
|
||||||
|
#' than revenue (series breaks `SB197`-`SB202`).
|
||||||
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||||
#' `revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
#' `revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
||||||
#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
||||||
@@ -17,6 +42,7 @@
|
|||||||
cog_revenue <- function(govid, years, category = NULL,
|
cog_revenue <- function(govid, years, category = NULL,
|
||||||
per_capita = FALSE, adjust_to_year = NULL,
|
per_capita = FALSE, adjust_to_year = NULL,
|
||||||
basis = c("harmonized", "raw"), recipe = NULL,
|
basis = c("harmonized", "raw"), recipe = NULL,
|
||||||
|
revenue_concept = c("general", "total"),
|
||||||
complete = FALSE) {
|
complete = FALSE) {
|
||||||
# flow_prefixes no longer classifies rows (crosswalk revenue_subtype
|
# flow_prefixes no longer classifies rows (crosswalk revenue_subtype
|
||||||
# membership does -- General Revenue, i.e. everything except
|
# membership does -- General Revenue, i.e. everything except
|
||||||
@@ -35,6 +61,7 @@ cog_revenue <- function(govid, years, category = NULL,
|
|||||||
adjust_to_year = adjust_to_year,
|
adjust_to_year = adjust_to_year,
|
||||||
basis = basis,
|
basis = basis,
|
||||||
recipe = recipe,
|
recipe = recipe,
|
||||||
|
revenue_concept = revenue_concept,
|
||||||
complete = complete
|
complete = complete
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|||||||
+37
-7
@@ -31,9 +31,26 @@
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
# cog_revenue()'s single concept (until uscogdata#12 adds more): Census
|
# The two revenue concepts (uscogdata#12), again as crosswalk subtype sets.
|
||||||
# General Revenue -- every crosswalk revenue subtype except insurance_trust.
|
# Census's manual section 4.3 defines the first by SUBTRACTING from the second
|
||||||
|
# -- "General revenue comprises all revenue except that classified as liquor
|
||||||
|
# store, utility, or insurance trust revenue" -- giving the identity
|
||||||
|
#
|
||||||
|
# Total Revenue = General + Utility + Liquor Store + Insurance Trust
|
||||||
|
#
|
||||||
|
# Verified against Census's own computed concept fields (IndFin FY2012,
|
||||||
|
# Wisconsin state): 31,410,686 + 0 + 0 + 4,469,906 = 35,880,592, exact.
|
||||||
.revenue_subtypes_general <- c("own_source", "federal", "state", "local_aid")
|
.revenue_subtypes_general <- c("own_source", "federal", "state", "local_aid")
|
||||||
|
.revenue_subtypes_total <- c(.revenue_subtypes_general, "utility",
|
||||||
|
"liquor_store", "insurance_trust")
|
||||||
|
|
||||||
|
#' @noRd
|
||||||
|
.revenue_concept_subtypes <- function(concept) {
|
||||||
|
switch(concept,
|
||||||
|
general = .revenue_subtypes_general,
|
||||||
|
total = .revenue_subtypes_total
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
#' Summarized spending by category
|
#' Summarized spending by category
|
||||||
#'
|
#'
|
||||||
@@ -198,6 +215,7 @@ cog_spending <- function(govid, years, category = NULL,
|
|||||||
per_capita, adjust_to_year,
|
per_capita, adjust_to_year,
|
||||||
basis = c("harmonized", "raw"), recipe = NULL,
|
basis = c("harmonized", "raw"), recipe = NULL,
|
||||||
expenditure_concept = c("primary", "direct", "total"),
|
expenditure_concept = c("primary", "direct", "total"),
|
||||||
|
revenue_concept = c("general", "total"),
|
||||||
complete = FALSE) {
|
complete = FALSE) {
|
||||||
basis_explicit <- length(basis) == 1L
|
basis_explicit <- length(basis) == 1L
|
||||||
basis <- match.arg(basis, c("harmonized", "raw"))
|
basis <- match.arg(basis, c("harmonized", "raw"))
|
||||||
@@ -215,16 +233,27 @@ cog_spending <- function(govid, years, category = NULL,
|
|||||||
}
|
}
|
||||||
)
|
)
|
||||||
|
|
||||||
|
revenue_concept <- tryCatch(
|
||||||
|
match.arg(revenue_concept, c("general", "total")),
|
||||||
|
error = function(e) {
|
||||||
|
cli::cli_abort(
|
||||||
|
"`revenue_concept` must be one of {.val general} or {.val total}.",
|
||||||
|
class = "uscogdata_invalid_revenue_concept",
|
||||||
|
parent = e
|
||||||
|
)
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
# The concept's subtype scope. Every code path below -- the verb SQL, the
|
# The concept's subtype scope. Every code path below -- the verb SQL, the
|
||||||
# harmonization exclusion count, and the complete = TRUE grid -- is scoped
|
# harmonization exclusion count, and the complete = TRUE grid -- is scoped
|
||||||
# by crosswalk subtype membership, never by item-code prefix. For revenue
|
# by crosswalk subtype membership, never by item-code prefix. The
|
||||||
# there is a single concept today (General Revenue; uscogdata#12 will add
|
# expenditure "total" concept's extra intergovernmental leg is the one
|
||||||
# more). "total"'s extra intergovernmental leg travels through the ig_*
|
# exception: it travels through the ig_* views rather than this scope,
|
||||||
# views, not through this scope.
|
# because its legacy rows are aggregate-flagged.
|
||||||
subtype_scope <- if (identical(subtype_col, "spend_subtype")) {
|
subtype_scope <- if (identical(subtype_col, "spend_subtype")) {
|
||||||
.expenditure_concept_subtypes(expenditure_concept)
|
.expenditure_concept_subtypes(expenditure_concept)
|
||||||
} else {
|
} else {
|
||||||
.revenue_subtypes_general
|
.revenue_concept_subtypes(revenue_concept)
|
||||||
}
|
}
|
||||||
|
|
||||||
govid <- .coerce_govid_input(govid, arg = "govid")
|
govid <- .coerce_govid_input(govid, arg = "govid")
|
||||||
@@ -427,6 +456,7 @@ cog_spending <- function(govid, years, category = NULL,
|
|||||||
expenditure_concept = expenditure_concept,
|
expenditure_concept = expenditure_concept,
|
||||||
expenditure_concept_note = expenditure_concept_note_for_prov,
|
expenditure_concept_note = expenditure_concept_note_for_prov,
|
||||||
expenditure_concept_direct_suppressed = direct_suppressed_flag,
|
expenditure_concept_direct_suppressed = direct_suppressed_flag,
|
||||||
|
revenue_concept = revenue_concept,
|
||||||
harmonization = harmonization,
|
harmonization = harmonization,
|
||||||
recipe = recipe_block,
|
recipe = recipe_block,
|
||||||
suggestions = suggestions,
|
suggestions = suggestions,
|
||||||
|
|||||||
@@ -71,6 +71,38 @@ enforce this by refusing `expenditure_concept = "total"`. See
|
|||||||
`vignette("total-spending", package = "uscogdata")` for the full
|
`vignette("total-spending", package = "uscogdata")` for the full
|
||||||
explanation with worked examples.
|
explanation with worked examples.
|
||||||
|
|
||||||
|
## General vs Total revenue
|
||||||
|
|
||||||
|
`cog_revenue(..., revenue_concept = c("general", "total"))` selects between
|
||||||
|
Census's two published revenue concepts, again defined as crosswalk
|
||||||
|
`revenue_subtype` sets rather than item-code prefixes:
|
||||||
|
|
||||||
|
- `"general"` (the default) is Census **General Revenue**: own-source
|
||||||
|
(taxes, charges, miscellaneous) plus federal, state and local
|
||||||
|
intergovernmental aid.
|
||||||
|
- `"total"` is Census **Total Revenue**: `general` plus utility revenue
|
||||||
|
(`A91`–`A94`), liquor store revenue (`A90`), and insurance trust revenue
|
||||||
|
(unemployment and workers' compensation `Y` codes plus the
|
||||||
|
employee-retirement `X` codes).
|
||||||
|
|
||||||
|
The manual defines the first by subtracting the other three from the second,
|
||||||
|
so the two are related by Census's own identity:
|
||||||
|
|
||||||
|
```
|
||||||
|
Total Revenue = General + Utility + Liquor Store + Insurance Trust
|
||||||
|
```
|
||||||
|
|
||||||
|
Two things worth knowing before switching to `"total"`:
|
||||||
|
|
||||||
|
- **Utility revenue is large for cities.** Measured on the bundled fixture,
|
||||||
|
utility plus liquor store revenue is 15.9% of city (type 2) revenue, versus
|
||||||
|
1.2% for states and 1.7% for counties. `general` excludes it by definition.
|
||||||
|
- **The employee-retirement (`X`) codes stop at FY2016**, when those systems
|
||||||
|
moved out of the annual finance file into the separate Annual Survey of
|
||||||
|
Public Pensions. A `"total"` series therefore steps down at the
|
||||||
|
FY2016/FY2017 seam for reasons of collection scope, not revenue (series
|
||||||
|
breaks `SB197`–`SB202`, in the corpus's `series_breaks` table).
|
||||||
|
|
||||||
## Developer notes
|
## Developer notes
|
||||||
|
|
||||||
### Testing
|
### Testing
|
||||||
|
|||||||
Binary file not shown.
Binary file not shown.
+4
-4
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"schema_version": 6,
|
"schema_version": 6,
|
||||||
"built_at": "2026-07-30T20:01:56Z",
|
"built_at": "2026-07-31T00:47:27Z",
|
||||||
"pipeline_commit": "e64a046",
|
"pipeline_commit": "aadb46b",
|
||||||
"fixture_note": "Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated from the sparsified schema-v6 corpus: the wide era (<= FY2011) no longer stores explicit zeros, so FY2011 absence means Census published $0 while FY2012+ absence means not reported. representation.parquet and code_set.parquet carry that rule and ship in full, as do every other metadata table in the publish tree. 2011/2012 straddle both the wide-aggregate -> modern-leaf format boundary (exercised by basis=\"harmonized\" and recipe= queries) and the dense -> sparse representation boundary (SB194); 2019/2020 retain the prior per-capita/CPI regression anchors. Regenerated via data-raw/regenerate_fixture_corpus.R.",
|
"fixture_note": "Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated from the sparsified schema-v6 corpus: the wide era (<= FY2011) no longer stores explicit zeros, so FY2011 absence means Census published $0 while FY2012+ absence means not reported. representation.parquet and code_set.parquet carry that rule and ship in full, as do every other metadata table in the publish tree. 2011/2012 straddle both the wide-aggregate -> modern-leaf format boundary (exercised by basis=\"harmonized\" and recipe= queries) and the dense -> sparse representation boundary (SB194); 2019/2020 retain the prior per-capita/CPI regression anchors. Regenerated via data-raw/regenerate_fixture_corpus.R.",
|
||||||
"data_vintage": {
|
"data_vintage": {
|
||||||
"source_vintages": {
|
"source_vintages": {
|
||||||
@@ -105,12 +105,12 @@
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
"path": "data/series_breaks.parquet",
|
"path": "data/series_breaks.parquet",
|
||||||
"sha256": "381090660c8b8a1bee852e7f63d29b9ecaf10f71870f41c63de92017c83b6f1f",
|
"sha256": "06dcc995ff533e57cc65fa25086cc9bf83ba592c58bf7cc99269dc2576f69944",
|
||||||
"description": "series_breaks.parquet"
|
"description": "series_breaks.parquet"
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"path": "data/summary_categories.parquet",
|
"path": "data/summary_categories.parquet",
|
||||||
"sha256": "e4918abf8e9e6d1372d7ddc255dc199f9c303e250f4106449241b48c68abee67",
|
"sha256": "e3b0efa00ce713b8f45829b89cfde24b55333f26101f0495df82d85997d18d8e",
|
||||||
"description": "summary_categories.parquet"
|
"description": "summary_categories.parquet"
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
|
|||||||
@@ -25,6 +25,12 @@
|
|||||||
"type": "boolean",
|
"type": "boolean",
|
||||||
"description": "TRUE when expenditure_concept = 'total' and at least one requested (year, category) has intergovernmental rows but NO Direct rows in this corpus (typically a legacy aggregate-only family) -- those result rows report the intergovernmental leg alone, not Direct + IG. Always FALSE for expenditure_concept = 'primary' or 'direct'. See the affected rows' `notes` for the recovering recipe, if any."
|
"description": "TRUE when expenditure_concept = 'total' and at least one requested (year, category) has intergovernmental rows but NO Direct rows in this corpus (typically a legacy aggregate-only family) -- those result rows report the intergovernmental leg alone, not Direct + IG. Always FALSE for expenditure_concept = 'primary' or 'direct'. See the affected rows' `notes` for the recovering recipe, if any."
|
||||||
},
|
},
|
||||||
|
"revenue_concept": {
|
||||||
|
"type": "string",
|
||||||
|
"enum": ["general", "total"],
|
||||||
|
"description": "Which revenue concept produced this result, defined as crosswalk revenue_subtype sets (never item-code prefixes). 'general' (the default) is Census General Revenue: own_source + federal + state + local_aid. 'total' is Census Total Revenue: general plus utility, liquor store and insurance trust revenue. Census defines the first by subtracting the other three from the second (manual section 4.3). Meaningful for cog_revenue() results; spending results carry the default.",
|
||||||
|
"$comment": "The employee-retirement (X) codes inside insurance_trust stop at FY2016, so a 'total' series steps at the FY2016/FY2017 seam for collection-scope reasons (series breaks SB197-SB202)."
|
||||||
|
},
|
||||||
"harmonization": { "type": "object" },
|
"harmonization": { "type": "object" },
|
||||||
"recipe": { "type": ["object", "null"] },
|
"recipe": { "type": ["object", "null"] },
|
||||||
"suggestions": { "type": "array" },
|
"suggestions": { "type": "array" },
|
||||||
|
|||||||
@@ -1,15 +1,18 @@
|
|||||||
-- Revenue rows, classified by crosswalk MEMBERSHIP rather than item-code
|
-- Revenue rows, classified by crosswalk MEMBERSHIP rather than item-code
|
||||||
-- first letter (see 20-spending_long.sql for why prefixes cannot work).
|
-- first letter (see 20-spending_long.sql for why prefixes cannot work).
|
||||||
-- Scope is Census General Revenue: every crosswalk revenue subtype EXCEPT
|
--
|
||||||
-- insurance_trust (Y01/Y02/Y04/Y11/Y12/Y51/Y52). Owner ruling 2026-07-30:
|
-- Carries EVERY revenue subtype. Which of Census's two published concepts a
|
||||||
-- the default revenue concept stays general; surfacing insurance-trust
|
-- query actually returns is decided per revenue_concept in R
|
||||||
-- revenue through an explicit concept argument is uscogdata#12.
|
-- (.verb_spendrev), exactly as expenditure_concept narrows spending_long:
|
||||||
|
-- general = own_source + federal + state + local_aid (the default)
|
||||||
|
-- total = general + utility + liquor_store + insurance_trust
|
||||||
|
-- Census defines the first by subtracting the other three from the second
|
||||||
|
-- (manual section 4.3), so both concepts need all four families present here.
|
||||||
CREATE OR REPLACE VIEW revenue_long AS
|
CREATE OR REPLACE VIEW revenue_long AS
|
||||||
SELECT *
|
SELECT *
|
||||||
FROM long
|
FROM long
|
||||||
WHERE item_code IN (
|
WHERE item_code IN (
|
||||||
SELECT item_code FROM summary_categories
|
SELECT item_code FROM summary_categories
|
||||||
WHERE category_type = 'revenue'
|
WHERE category_type = 'revenue'
|
||||||
AND revenue_subtype <> 'insurance_trust'
|
|
||||||
)
|
)
|
||||||
AND NOT is_aggregate;
|
AND NOT is_aggregate;
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
-- Harmonized-basis twin of 21-revenue_long.sql: same crosswalk-membership
|
-- Harmonized-basis twin of 21-revenue_long.sql: same crosswalk-membership
|
||||||
-- classification (General Revenue = revenue minus insurance_trust), applied
|
-- classification (every revenue subtype; the concept narrows in R), applied
|
||||||
-- to harmonized_code rather than the published item_code.
|
-- to harmonized_code rather than the published item_code.
|
||||||
CREATE OR REPLACE VIEW revenue_long_harmonized AS
|
CREATE OR REPLACE VIEW revenue_long_harmonized AS
|
||||||
SELECT * REPLACE (harmonized_code AS item_code)
|
SELECT * REPLACE (harmonized_code AS item_code)
|
||||||
@@ -9,5 +9,4 @@ WHERE NOT is_aggregate
|
|||||||
AND harmonized_code IN (
|
AND harmonized_code IN (
|
||||||
SELECT item_code FROM summary_categories
|
SELECT item_code FROM summary_categories
|
||||||
WHERE category_type = 'revenue'
|
WHERE category_type = 'revenue'
|
||||||
AND revenue_subtype <> 'insurance_trust'
|
|
||||||
);
|
);
|
||||||
|
|||||||
@@ -12,6 +12,7 @@ cog_revenue(
|
|||||||
adjust_to_year = NULL,
|
adjust_to_year = NULL,
|
||||||
basis = c("harmonized", "raw"),
|
basis = c("harmonized", "raw"),
|
||||||
recipe = NULL,
|
recipe = NULL,
|
||||||
|
revenue_concept = c("general", "total"),
|
||||||
complete = FALSE
|
complete = FALSE
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
@@ -57,6 +58,32 @@ argument is ignored and the result's provenance reports
|
|||||||
FALSE`, pointing at the `recipe` block instead) rather than a
|
FALSE`, pointing at the `recipe` block instead) rather than a
|
||||||
possibly-misleading `"harmonized"`/`"raw"` value.}
|
possibly-misleading `"harmonized"`/`"raw"` value.}
|
||||||
|
|
||||||
|
\item{revenue_concept}{Which of Census's two published revenue concepts to
|
||||||
|
return. Concepts are defined as sets of the crosswalk's `revenue_subtype`
|
||||||
|
values -- never as item-code first letters, which cannot classify
|
||||||
|
correctly (prefix `Y` spans revenue, expenditure and balance codes, and
|
||||||
|
prefix `X` does the same):
|
||||||
|
|
||||||
|
* `"general"` (default) -- Census General Revenue: `own_source` +
|
||||||
|
`federal` + `state` + `local_aid`. The manual defines this concept by
|
||||||
|
subtraction (section 4.3: *"General revenue comprises all revenue
|
||||||
|
except that classified as liquor store, utility, or insurance trust
|
||||||
|
revenue"*), so utility (`A91`-`A94`), liquor store (`A90`) and
|
||||||
|
insurance trust revenue are all excluded.
|
||||||
|
* `"total"` -- Census Total Revenue: every revenue subtype, i.e.
|
||||||
|
`general` plus utility, liquor store, and insurance trust revenue
|
||||||
|
(`Y01`/`Y02`/`Y04`/`Y11`/`Y12`/`Y51`/`Y52` and the employee-retirement
|
||||||
|
`X01`/`X02`/`X05`/`X08`).
|
||||||
|
|
||||||
|
The two are related by Census's own identity, `Total Revenue = General +
|
||||||
|
Utility + Liquor Store + Insurance Trust`.
|
||||||
|
|
||||||
|
Note that the employee-retirement (`X`) codes stop at FY2016, when those
|
||||||
|
systems moved out of the annual finance file into the separate Annual
|
||||||
|
Survey of Public Pensions, so a `"total"` series steps down at the
|
||||||
|
FY2016/FY2017 seam for reasons that are about collection scope rather
|
||||||
|
than revenue (series breaks `SB197`-`SB202`).}
|
||||||
|
|
||||||
\item{complete}{If `TRUE`, fill the requested grid so that a cell the
|
\item{complete}{If `TRUE`, fill the requested grid so that a cell the
|
||||||
corpus does not carry still appears, labelled with **why** it is
|
corpus does not carry still appears, labelled with **why** it is
|
||||||
missing, and add a `value_source` column to every row:
|
missing, and add a `value_source` column to every row:
|
||||||
|
|||||||
@@ -49,12 +49,14 @@ test_that("cog_categories(type = 'revenue') returns only revenue rows", {
|
|||||||
skip_if_no_corpus()
|
skip_if_no_corpus()
|
||||||
r <- cog_categories(type = "revenue")
|
r <- cog_categories(type = "revenue")
|
||||||
expect_true(all(r$category_type == "revenue"))
|
expect_true(all(r$category_type == "revenue"))
|
||||||
# `insurance_trust` (Y01/Y02/Y04/Y11/Y12/Y51/Y52) is deliberately NOT
|
# The four non-general subtypes are deliberately NOT own_source: Census's
|
||||||
# own_source: Census's "General Revenue" excludes insurance trust revenue,
|
# General Revenue excludes insurance trust (Y01 alone is $1.31T corpus-wide,
|
||||||
# and Y01 alone is $1.31T corpus-wide.
|
# plus the employee-retirement X codes), utility (A91-A94) and liquor store
|
||||||
|
# (A90) revenue by definition, which is what makes both of its published
|
||||||
|
# revenue concepts computable -- see `revenue_concept` in `?cog_revenue`.
|
||||||
expect_true(all(r$subtype %in%
|
expect_true(all(r$subtype %in%
|
||||||
c("own_source", "federal", "state", "local_aid",
|
c("own_source", "federal", "state", "local_aid",
|
||||||
"insurance_trust")))
|
"insurance_trust", "utility", "liquor_store")))
|
||||||
})
|
})
|
||||||
|
|
||||||
test_that("cog_categories(pattern = ...) filters case-insensitively", {
|
test_that("cog_categories(pattern = ...) filters case-insensitively", {
|
||||||
|
|||||||
@@ -9,47 +9,131 @@
|
|||||||
# a published Census revenue concept exactly the way I89 sits inside Census's
|
# a published Census revenue concept exactly the way I89 sits inside Census's
|
||||||
# Direct Expenditure concept (finding F-012).
|
# Direct Expenditure concept (finding F-012).
|
||||||
#
|
#
|
||||||
# CAVEAT FOR WHOEVER PICKS THIS UP: the argument name below (`revenue_concept =
|
# RULED 2026-07-30. `revenue_concept = c("general", "total")` mirrors
|
||||||
# "total"`) is this test's *proposal*, not a settled decision. The owner's
|
# `expenditure_concept`, and the two values are Census's two published revenue
|
||||||
# 2026-07-28 resolution covers expenditure concepts only; no revenue-side
|
# concepts, related by the manual's own identity (section 4.3, which defines
|
||||||
# naming has been ruled on. If the eventual argument is named differently,
|
# the first by SUBTRACTING from the second):
|
||||||
# change the two calls here -- the asserted dollar invariants are what matter
|
#
|
||||||
# and are independent of the naming.
|
# Total Revenue = General + Utility + Liquor Store + Insurance Trust
|
||||||
|
#
|
||||||
|
# so `general` is the four general subtypes (own_source/federal/state/
|
||||||
|
# local_aid) and `total` is every revenue subtype. Naming utility (A91-A94)
|
||||||
|
# and liquor store (A90) separately is what makes BOTH computable -- before
|
||||||
|
# cog_pipeline#79 they sat in own_source, so the default was really
|
||||||
|
# "General + Utility + Liquor", a concept Census does not publish.
|
||||||
#
|
#
|
||||||
# Fixture reproducibility: Madison's own X-prefix revenue (FY1970-FY1986,
|
# Fixture reproducibility: Madison's own X-prefix revenue (FY1970-FY1986,
|
||||||
# $15,098,000 nominal, $0 thereafter) is outside the bundled fixture's year
|
# $15,098,000 nominal, $0 thereafter) is outside the bundled fixture's year
|
||||||
# window (2011/2012/2019/2020), so the same invariant is asserted on Wisconsin
|
# window (2011/2012/2019/2020), so the same invariant is asserted on Wisconsin
|
||||||
# state government FY2012, where the fixture carries nonzero X01/X05/X08.
|
# state government FY2012, where the fixture carries nonzero X01/X02/X05/X08.
|
||||||
|
|
||||||
test_that("cog_revenue() can return Census Total Revenue including Insurance Trust (prefix X)", {
|
test_that("cog_revenue() can return Census Total Revenue including Insurance Trust (prefix X)", {
|
||||||
testthat::skip("Blocked on uscogdata#12 (finding F-014)")
|
|
||||||
|
|
||||||
wi_state <- "550000227544" # WISCONSIN (state government)
|
wi_state <- "550000227544" # WISCONSIN (state government)
|
||||||
|
|
||||||
# Revenue-shaped Employee Retirement codes, read from the RAW corpus rather
|
# Revenue-shaped Employee Retirement codes, read from the RAW corpus rather
|
||||||
# than through cog_revenue(), which is the filter under test:
|
# than through cog_revenue(), which is the filter under test:
|
||||||
# X01 local employee contribution, X04/X05 contributions and transfers from
|
# X01/X02 employee contributions, X05 contributions from other governments,
|
||||||
# other governments, X08 earnings on investments.
|
# X08 total earnings on investments.
|
||||||
x_revenue <- wt_raw_amt(wi_state, 2012L, codes = c("X01", "X04", "X05", "X08"))
|
#
|
||||||
expect_equal(x_revenue, 2038800) # 615,835 + 0 + 560,382 + 862,583 ($1,000s)
|
# X04 and X06 are deliberately NOT in this set, though an earlier draft of
|
||||||
|
# this test included X04. Both are exhibit codes for INTRAgovernmental
|
||||||
|
# transfers (the administering government paying into its own fund), which
|
||||||
|
# X05's own definition excludes by name. Census agrees: its computed "Total
|
||||||
|
# Emp Ret Rev" for this government-year is exactly the four codes below.
|
||||||
|
x_revenue <- wt_raw_amt(wi_state, 2012L, codes = c("X01", "X02", "X05", "X08"))
|
||||||
|
expect_equal(x_revenue, 2283883) # 615,835 + 245,083 + 560,382 + 862,583
|
||||||
|
|
||||||
|
# The Y-prefix insurance trust revenue (unemployment + workers comp), which
|
||||||
|
# is the other half of the same Census concept.
|
||||||
|
y_revenue <- wt_raw_amt(wi_state, 2012L, codes = c("Y01", "Y11"))
|
||||||
|
expect_equal(y_revenue, 1259785)
|
||||||
|
|
||||||
general <- cog_revenue(govid = wi_state, years = 2012L)
|
general <- cog_revenue(govid = wi_state, years = 2012L)
|
||||||
|
expect_equal(attr(general, "provenance")$revenue_concept, "general")
|
||||||
expect_equal(sum(general$amt_nominal), 31338293000)
|
expect_equal(sum(general$amt_nominal), 31338293000)
|
||||||
|
|
||||||
total <- cog_revenue(govid = wi_state, years = 2012L, revenue_concept = "total")
|
total <- cog_revenue(govid = wi_state, years = 2012L, revenue_concept = "total")
|
||||||
expect_equal(sum(total$amt_nominal) - sum(general$amt_nominal), x_revenue * 1000)
|
expect_equal(attr(total, "provenance")$revenue_concept, "total")
|
||||||
expect_equal(sum(total$amt_nominal), 33377093000)
|
|
||||||
expect_true(all(c("X01", "X05", "X08") %in% wt_codes_included(total)))
|
# total - general is the whole insurance trust leg, X and Y together.
|
||||||
|
# Asserted as a delta as well as a level so this stays correct however the
|
||||||
|
# utility/liquor families land (both are $0 for WI state in FY2012).
|
||||||
|
expect_equal(sum(total$amt_nominal) - sum(general$amt_nominal),
|
||||||
|
(x_revenue + y_revenue) * 1000)
|
||||||
|
expect_equal(sum(total$amt_nominal), 34881961000)
|
||||||
|
expect_true(all(c("X01", "X02", "X05", "X08") %in% wt_codes_included(total)))
|
||||||
|
|
||||||
# Sibling codes under the SAME first letter must stay out: X11/X12 are
|
# Sibling codes under the SAME first letter must stay out: X11/X12 are
|
||||||
# benefit payments (an expenditure) and X21/X30/X47 are cash and securities
|
# benefit payments (an expenditure) and X21/X30/X47 are cash and securities
|
||||||
# holdings (a balance-sheet stock). This is the F-018 point restated on the
|
# holdings (a balance-sheet stock). This is the F-018 point restated on the
|
||||||
# revenue side -- the split has to come from the crosswalk's spend_type, not
|
# revenue side -- the split comes from the crosswalk, not from the letter X.
|
||||||
# from the letter X.
|
|
||||||
expect_false(any(c("X11", "X12", "X21", "X30", "X47") %in% wt_codes_included(total)))
|
expect_false(any(c("X11", "X12", "X21", "X30", "X47") %in% wt_codes_included(total)))
|
||||||
|
|
||||||
# Every returned row still resolves to a category. summary_categories has
|
# Every returned row still resolves to a category (cog_pipeline#79 added the
|
||||||
# zero rows for prefix X today, so relaxing the prefix filter alone would
|
# X crosswalk rows; relaxing a prefix filter alone would have produced
|
||||||
# produce category = NA rows -- see census_of_governments_finance_pipeline#60.
|
# category = NA rows).
|
||||||
expect_false(any(is.na(total$category)))
|
expect_false(any(is.na(total$category)))
|
||||||
})
|
})
|
||||||
|
|
||||||
|
test_that("revenue_concept = 'general' is the default and is strict Census General Revenue", {
|
||||||
|
wi_state <- "550000227544"
|
||||||
|
default <- cog_revenue(govid = wi_state, years = 2012L)
|
||||||
|
explicit <- cog_revenue(govid = wi_state, years = 2012L,
|
||||||
|
revenue_concept = "general")
|
||||||
|
expect_equal(sum(default$amt_nominal), sum(explicit$amt_nominal))
|
||||||
|
|
||||||
|
# General Revenue excludes utility, liquor store AND insurance trust
|
||||||
|
# revenue. WI state carries $0 of utility/liquor in FY2012, so the level
|
||||||
|
# assertion above cannot see those two -- assert the subtype scope directly.
|
||||||
|
#
|
||||||
|
# A subset, not setequal: `state` means "intergovernmental revenue FROM the
|
||||||
|
# state government" (the C codes), which a STATE government does not receive
|
||||||
|
# from itself, so it is legitimately absent here.
|
||||||
|
expect_true(all(default$revenue_subtype %in%
|
||||||
|
c("own_source", "federal", "state", "local_aid")))
|
||||||
|
expect_false(any(c("utility", "liquor_store", "insurance_trust") %in%
|
||||||
|
default$revenue_subtype))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("utility and liquor store revenue are inside `total` and outside `general`", {
|
||||||
|
# A city, where utility revenue is material: this is the case the WI state
|
||||||
|
# baseline structurally cannot exercise. Measured on the fixture, utility +
|
||||||
|
# liquor is 15.9% of what cog_revenue() returned for type-2 governments
|
||||||
|
# before the general/total split, so this is the largest behaviour change
|
||||||
|
# the concept split introduces.
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
gov <- DBI::dbGetQuery(con,
|
||||||
|
"SELECT canonical_govid, SUM(amt) amt FROM long
|
||||||
|
WHERE year = 2012 AND type = 2 AND NOT is_aggregate
|
||||||
|
AND item_code IN ('A91','A92','A93','A94')
|
||||||
|
GROUP BY 1 ORDER BY amt DESC LIMIT 1")$canonical_govid
|
||||||
|
|
||||||
|
util_raw <- wt_raw_amt(gov, 2012L, codes = c("A90", "A91", "A92", "A93", "A94"))
|
||||||
|
expect_gt(util_raw, 0)
|
||||||
|
|
||||||
|
general <- cog_revenue(govid = gov, years = 2012L)
|
||||||
|
total <- cog_revenue(govid = gov, years = 2012L, revenue_concept = "total")
|
||||||
|
|
||||||
|
expect_false(any(c("utility", "liquor_store") %in% general$revenue_subtype))
|
||||||
|
expect_true("utility" %in% total$revenue_subtype)
|
||||||
|
expect_equal(sum(total$amt_nominal) - sum(general$amt_nominal),
|
||||||
|
util_raw * 1000 +
|
||||||
|
wt_raw_amt(gov, 2012L, codes = c("Y01", "Y11", "X01", "X02",
|
||||||
|
"X05", "X08")) * 1000)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("revenue_concept rejects unknown values and never returns a balance row", {
|
||||||
|
expect_error(
|
||||||
|
cog_revenue("550000227544", years = 2012L, revenue_concept = "gross"),
|
||||||
|
class = "uscogdata_invalid_revenue_concept"
|
||||||
|
)
|
||||||
|
|
||||||
|
# uscogdata#25 restated for the widest revenue concept: stocks are not
|
||||||
|
# flows, and `total` must not quietly admit the X/Y/W/Z balance families.
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
balance <- DBI::dbGetQuery(con,
|
||||||
|
"SELECT item_code, category FROM summary_categories WHERE category_type = 'balance'")
|
||||||
|
total <- cog_revenue("550000227544", years = 2012L, revenue_concept = "total")
|
||||||
|
expect_false(any(total$category %in% balance$category))
|
||||||
|
expect_length(intersect(wt_codes_included(total), balance$item_code), 0L)
|
||||||
|
})
|
||||||
|
|||||||
@@ -393,19 +393,19 @@ test_that("spending_long carries exactly the non-IG expenditure crosswalk codes
|
|||||||
expect_equal(agg_count, 0)
|
expect_equal(agg_count, 0)
|
||||||
})
|
})
|
||||||
|
|
||||||
test_that("revenue_long carries exactly the general-revenue crosswalk codes and excludes aggregates", {
|
test_that("revenue_long carries exactly the revenue crosswalk codes and excludes aggregates", {
|
||||||
skip_if_no_corpus()
|
skip_if_no_corpus()
|
||||||
con <- cog_open()
|
con <- cog_open()
|
||||||
on.exit(cog_close())
|
on.exit(cog_close())
|
||||||
|
|
||||||
# General Revenue scope: revenue crosswalk members minus insurance_trust
|
# The view carries EVERY revenue subtype; which of Census's two published
|
||||||
# (owner ruling 2026-07-30; an explicit wider concept is uscogdata#12).
|
# concepts a query returns is decided per `revenue_concept` in R
|
||||||
|
# (uscogdata#12), exactly as `expenditure_concept` narrows spending_long.
|
||||||
stray <- DBI::dbGetQuery(con,
|
stray <- DBI::dbGetQuery(con,
|
||||||
"SELECT DISTINCT s.item_code
|
"SELECT DISTINCT s.item_code
|
||||||
FROM revenue_long s
|
FROM revenue_long s
|
||||||
LEFT JOIN summary_categories c USING (item_code)
|
LEFT JOIN summary_categories c USING (item_code)
|
||||||
WHERE c.category_type IS DISTINCT FROM 'revenue'
|
WHERE c.category_type IS DISTINCT FROM 'revenue'"
|
||||||
OR c.revenue_subtype = 'insurance_trust'"
|
|
||||||
)$item_code
|
)$item_code
|
||||||
expect_length(stray, 0L)
|
expect_length(stray, 0L)
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user