diff --git a/DESCRIPTION b/DESCRIPTION index 7e248c8..7a45864 100644 --- a/DESCRIPTION +++ b/DESCRIPTION @@ -22,6 +22,7 @@ Imports: httr2, digest Suggests: + arrow, testthat (>= 3.0.0), withr, knitr, diff --git a/NAMESPACE b/NAMESPACE index 1c67f0b..5d956a0 100644 --- a/NAMESPACE +++ b/NAMESPACE @@ -1,5 +1,6 @@ # Generated by roxygen2: do not edit by hand +export(cog_balances) export(cog_basket_resolution) export(cog_basket_unresolved) export(cog_categories) diff --git a/R/balances.R b/R/balances.R new file mode 100644 index 0000000..fd8b4ee --- /dev/null +++ b/R/balances.R @@ -0,0 +1,98 @@ +# R/balances.R +# +# Cash and security holdings. A third verb rather than an argument on a money +# verb because holdings are a STOCK -- a balance at a point in time -- while +# cog_spending()/cog_revenue() return FLOWS over a fiscal year. The money +# verbs' whole argument vocabulary (expenditure_concept, revenue_concept, +# complete=) describes flows and is meaningless here, so this deliberately +# does NOT route through .verb_spendrev(). + +#' Cash and security holdings for one or more governments +#' +#' Returns Census cash-and-security holdings (`category_type = "balance"`): +#' fund balances, retirement system holdings and insurance trust balances. +#' +#' @section Holdings are not GAAP fund balance: +#' Census holdings are **gross** -- no liabilities are netted -- so a reserve +#' ratio built from them overstates what is actually available. They are not +#' comparable to a GAAP fund balance from an ACFR. +#' +#' @param govid Canonical govid(s): a character vector, or a data frame with a +#' `canonical_govid` column (e.g. from [cog_gov_search()]). +#' @param years Integer vector of fiscal years. +#' @param category Optional character vector of categories to keep. One of +#' `"Fund Balances"`, `"Insurance Trust Balances"`, +#' `"Retirement System Holdings"`. There is deliberately no `subtype` +#' argument: for holdings, `category` is a strict coarsening of +#' `balance_subtype` (unlike the money verbs, where the two axes cross), so +#' every combination would be either redundant or empty. +#' `category = "Fund Balances"` is exactly the `general` family +#' (`W01`/`W31`/`W61`). `balance_subtype` is returned, so a finer split is +#' one `dplyr::filter()` away. +#' @param per_capita Divide holdings by population. Note this is a **stock per +#' resident** (reserves per person), which is *not* comparable to +#' [cog_spending()]'s per-capita figures -- those are a flow per person. +#' @param adjust_to_year Deflate to this year's dollars (CPI-U). +#' @param basis Accepted for uniformity with the money verbs, but currently a +#' **no-op**: `harmonization_map` carries no balance-code rows, so harmonized +#' and raw space are identical for holdings. Reported in +#' `provenance$basis_note`. +#' @param recipe Optional harmonization recipe id (see [cog_recipes()]). +#' `"cash_securities_z77_wide"` and `"cash_securities_z78_wide"` bridge the +#' wide era to the modern one. +#' +#' @return A `tbl_df` with a `provenance` attribute. Amounts are full US +#' dollars. +#' @export +cog_balances <- function(govid, years, category = NULL, + per_capita = FALSE, adjust_to_year = NULL, + basis = c("harmonized", "raw"), recipe = NULL) { + call <- match.call() + basis <- match.arg(basis, c("harmonized", "raw")) + govid <- .coerce_govid_input(govid) + years <- as.integer(years) + if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year) + + con <- .ensure_session() + .require_balance_support(con) + .check_govids_in_scope(govid) + + basis_note <- paste0( + "`basis` has no effect on holdings: harmonization_map carries no ", + "balance-code rows, so harmonized and raw space are identical here." + ) + + sql <- .build_verb_sql("balance_annotated", "balance_subtype", + govid, years, category, + ig_view = NULL, subtype_scope = NULL) + result <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) + + prov <- .build_provenance( + verb = "cog_balances", call = call, govid = govid, years = years, + category = category, per_capita = per_capita, + adjust_to_year = adjust_to_year, result = result, sql = sql, + subtype_col = "balance_subtype", + basis = basis, basis_note = basis_note, + # Neither concept vocabulary applies to a stock. + expenditure_concept = NA_character_, + revenue_concept = NA_character_ + ) + + attr(result, "provenance") <- prov + result +} + +#' Abort unless the mounted corpus classifies balance codes. +#' +#' `balance_subtype` arrived with cog_pipeline #76/#77 without a +#' schema_version bump, so the check is on the column, not the version. +#' @noRd +.require_balance_support <- function(con) { + if (.corpus_has_balance_subtype(con)) return(invisible(TRUE)) + cli::cli_abort( + c("This corpus does not classify cash and security holdings.", + i = "`summary_categories` has no {.field balance_subtype} column.", + i = "Republish from cog_pipeline at #76/#77 or later."), + class = "uscogdata_no_balance_support" + ) +} diff --git a/man/cog_balances.Rd b/man/cog_balances.Rd new file mode 100644 index 0000000..81323a0 --- /dev/null +++ b/man/cog_balances.Rd @@ -0,0 +1,62 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/balances.R +\name{cog_balances} +\alias{cog_balances} +\title{Cash and security holdings for one or more governments} +\usage{ +cog_balances( + govid, + years, + category = NULL, + per_capita = FALSE, + adjust_to_year = NULL, + basis = c("harmonized", "raw"), + recipe = NULL +) +} +\arguments{ +\item{govid}{Canonical govid(s): a character vector, or a data frame with a +`canonical_govid` column (e.g. from [cog_gov_search()]).} + +\item{years}{Integer vector of fiscal years.} + +\item{category}{Optional character vector of categories to keep. One of +`"Fund Balances"`, `"Insurance Trust Balances"`, +`"Retirement System Holdings"`. There is deliberately no `subtype` +argument: for holdings, `category` is a strict coarsening of +`balance_subtype` (unlike the money verbs, where the two axes cross), so +every combination would be either redundant or empty. +`category = "Fund Balances"` is exactly the `general` family +(`W01`/`W31`/`W61`). `balance_subtype` is returned, so a finer split is +one `dplyr::filter()` away.} + +\item{per_capita}{Divide holdings by population. Note this is a **stock per +resident** (reserves per person), which is *not* comparable to +[cog_spending()]'s per-capita figures -- those are a flow per person.} + +\item{adjust_to_year}{Deflate to this year's dollars (CPI-U).} + +\item{basis}{Accepted for uniformity with the money verbs, but currently a +**no-op**: `harmonization_map` carries no balance-code rows, so harmonized +and raw space are identical for holdings. Reported in +`provenance$basis_note`.} + +\item{recipe}{Optional harmonization recipe id (see [cog_recipes()]). +`"cash_securities_z77_wide"` and `"cash_securities_z78_wide"` bridge the +wide era to the modern one.} +} +\value{ +A `tbl_df` with a `provenance` attribute. Amounts are full US + dollars. +} +\description{ +Returns Census cash-and-security holdings (`category_type = "balance"`): +fund balances, retirement system holdings and insurance trust balances. +} +\section{Holdings are not GAAP fund balance}{ + +Census holdings are **gross** -- no liabilities are netted -- so a reserve +ratio built from them overstates what is actually available. They are not +comparable to a GAAP fund balance from an ACFR. +} + diff --git a/tests/testthat/test-balances.R b/tests/testthat/test-balances.R index 5e5abb3..0af67b4 100644 --- a/tests/testthat/test-balances.R +++ b/tests/testthat/test-balances.R @@ -97,3 +97,60 @@ test_that("balance views are skipped on a corpus without balance_subtype", { expect_true("revenue_long" %in% views) }) }) + +test_that("cog_balances returns holdings for a government that has them", { + skip_if_no_corpus() + with_fixture_corpus({ + r <- cog_balances("550000227544", 2019) + expect_s3_class(r, "tbl_df") + expect_true(nrow(r) > 0L) + expect_true(all(c("year", "canonical_govid", "gov_name", "balance_subtype", + "category", "amt_nominal") %in% names(r))) + expect_identical(sort(unique(r$category)), + c("Fund Balances", "Insurance Trust Balances")) + expect_false(is.null(attr(r, "provenance"))) + expect_identical(attr(r, "provenance")$verb, "cog_balances") + }) +}) + +test_that('category = "Fund Balances" is exactly the general family', { + skip_if_no_corpus() + with_fixture_corpus({ + r <- cog_balances("550000227544", 2019, category = "Fund Balances") + expect_identical(unique(r$balance_subtype), "general") + codes <- sort(unlist(strsplit(paste(r$codes_included, collapse = ","), ","))) + expect_identical(codes, c("W01", "W31", "W61")) + }) +}) + +test_that("no flow code can reach cog_balances", { + skip_if_no_corpus() + with_fixture_corpus({ + r <- cog_balances("550000227544", c(2011, 2012, 2019, 2020)) + got <- unique(unlist(strsplit(paste(r$codes_included, collapse = ","), ","))) + + # The expected set is read from the RAW corpus, never from the verb -- + # verifying an absence through the filter that creates it proves nothing. + ds <- arrow::open_dataset(file.path(fixture_corpus_path(), "data", "long")) + sc <- arrow::read_parquet( + file.path(fixture_corpus_path(), "data", "summary_categories.parquet")) + sc <- as.data.frame(sc) + balance_codes <- sc$item_code[sc$category_type == "balance"] + + expect_true(all(got %in% balance_codes)) + expect_true(length(setdiff(got, balance_codes)) == 0L) + }) +}) + +test_that("every balance_subtype maps to exactly one category", { + skip_if_no_corpus() + # Dropping the `subtype` argument is only safe while this tree holds. If the + # pipeline ever gives a balance subtype a second category, `category` becomes + # a lossy filter -- fail HERE rather than in a user's analysis. + sc <- as.data.frame(arrow::read_parquet( + file.path(fixture_corpus_path(), "data", "summary_categories.parquet"))) + b <- sc[sc$category_type == "balance", ] + per_subtype <- tapply(b$category, b$balance_subtype, + function(x) length(unique(x))) + expect_true(all(per_subtype == 1L)) +})