feat: cog_balances() core verb (#25)
This commit is contained in:
@@ -22,6 +22,7 @@ Imports:
|
|||||||
httr2,
|
httr2,
|
||||||
digest
|
digest
|
||||||
Suggests:
|
Suggests:
|
||||||
|
arrow,
|
||||||
testthat (>= 3.0.0),
|
testthat (>= 3.0.0),
|
||||||
withr,
|
withr,
|
||||||
knitr,
|
knitr,
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
# Generated by roxygen2: do not edit by hand
|
# Generated by roxygen2: do not edit by hand
|
||||||
|
|
||||||
|
export(cog_balances)
|
||||||
export(cog_basket_resolution)
|
export(cog_basket_resolution)
|
||||||
export(cog_basket_unresolved)
|
export(cog_basket_unresolved)
|
||||||
export(cog_categories)
|
export(cog_categories)
|
||||||
|
|||||||
@@ -0,0 +1,98 @@
|
|||||||
|
# R/balances.R
|
||||||
|
#
|
||||||
|
# Cash and security holdings. A third verb rather than an argument on a money
|
||||||
|
# verb because holdings are a STOCK -- a balance at a point in time -- while
|
||||||
|
# cog_spending()/cog_revenue() return FLOWS over a fiscal year. The money
|
||||||
|
# verbs' whole argument vocabulary (expenditure_concept, revenue_concept,
|
||||||
|
# complete=) describes flows and is meaningless here, so this deliberately
|
||||||
|
# does NOT route through .verb_spendrev().
|
||||||
|
|
||||||
|
#' Cash and security holdings for one or more governments
|
||||||
|
#'
|
||||||
|
#' Returns Census cash-and-security holdings (`category_type = "balance"`):
|
||||||
|
#' fund balances, retirement system holdings and insurance trust balances.
|
||||||
|
#'
|
||||||
|
#' @section Holdings are not GAAP fund balance:
|
||||||
|
#' Census holdings are **gross** -- no liabilities are netted -- so a reserve
|
||||||
|
#' ratio built from them overstates what is actually available. They are not
|
||||||
|
#' comparable to a GAAP fund balance from an ACFR.
|
||||||
|
#'
|
||||||
|
#' @param govid Canonical govid(s): a character vector, or a data frame with a
|
||||||
|
#' `canonical_govid` column (e.g. from [cog_gov_search()]).
|
||||||
|
#' @param years Integer vector of fiscal years.
|
||||||
|
#' @param category Optional character vector of categories to keep. One of
|
||||||
|
#' `"Fund Balances"`, `"Insurance Trust Balances"`,
|
||||||
|
#' `"Retirement System Holdings"`. There is deliberately no `subtype`
|
||||||
|
#' argument: for holdings, `category` is a strict coarsening of
|
||||||
|
#' `balance_subtype` (unlike the money verbs, where the two axes cross), so
|
||||||
|
#' every combination would be either redundant or empty.
|
||||||
|
#' `category = "Fund Balances"` is exactly the `general` family
|
||||||
|
#' (`W01`/`W31`/`W61`). `balance_subtype` is returned, so a finer split is
|
||||||
|
#' one `dplyr::filter()` away.
|
||||||
|
#' @param per_capita Divide holdings by population. Note this is a **stock per
|
||||||
|
#' resident** (reserves per person), which is *not* comparable to
|
||||||
|
#' [cog_spending()]'s per-capita figures -- those are a flow per person.
|
||||||
|
#' @param adjust_to_year Deflate to this year's dollars (CPI-U).
|
||||||
|
#' @param basis Accepted for uniformity with the money verbs, but currently a
|
||||||
|
#' **no-op**: `harmonization_map` carries no balance-code rows, so harmonized
|
||||||
|
#' and raw space are identical for holdings. Reported in
|
||||||
|
#' `provenance$basis_note`.
|
||||||
|
#' @param recipe Optional harmonization recipe id (see [cog_recipes()]).
|
||||||
|
#' `"cash_securities_z77_wide"` and `"cash_securities_z78_wide"` bridge the
|
||||||
|
#' wide era to the modern one.
|
||||||
|
#'
|
||||||
|
#' @return A `tbl_df` with a `provenance` attribute. Amounts are full US
|
||||||
|
#' dollars.
|
||||||
|
#' @export
|
||||||
|
cog_balances <- function(govid, years, category = NULL,
|
||||||
|
per_capita = FALSE, adjust_to_year = NULL,
|
||||||
|
basis = c("harmonized", "raw"), recipe = NULL) {
|
||||||
|
call <- match.call()
|
||||||
|
basis <- match.arg(basis, c("harmonized", "raw"))
|
||||||
|
govid <- .coerce_govid_input(govid)
|
||||||
|
years <- as.integer(years)
|
||||||
|
if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year)
|
||||||
|
|
||||||
|
con <- .ensure_session()
|
||||||
|
.require_balance_support(con)
|
||||||
|
.check_govids_in_scope(govid)
|
||||||
|
|
||||||
|
basis_note <- paste0(
|
||||||
|
"`basis` has no effect on holdings: harmonization_map carries no ",
|
||||||
|
"balance-code rows, so harmonized and raw space are identical here."
|
||||||
|
)
|
||||||
|
|
||||||
|
sql <- .build_verb_sql("balance_annotated", "balance_subtype",
|
||||||
|
govid, years, category,
|
||||||
|
ig_view = NULL, subtype_scope = NULL)
|
||||||
|
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||||
|
|
||||||
|
prov <- .build_provenance(
|
||||||
|
verb = "cog_balances", call = call, govid = govid, years = years,
|
||||||
|
category = category, per_capita = per_capita,
|
||||||
|
adjust_to_year = adjust_to_year, result = result, sql = sql,
|
||||||
|
subtype_col = "balance_subtype",
|
||||||
|
basis = basis, basis_note = basis_note,
|
||||||
|
# Neither concept vocabulary applies to a stock.
|
||||||
|
expenditure_concept = NA_character_,
|
||||||
|
revenue_concept = NA_character_
|
||||||
|
)
|
||||||
|
|
||||||
|
attr(result, "provenance") <- prov
|
||||||
|
result
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Abort unless the mounted corpus classifies balance codes.
|
||||||
|
#'
|
||||||
|
#' `balance_subtype` arrived with cog_pipeline #76/#77 without a
|
||||||
|
#' schema_version bump, so the check is on the column, not the version.
|
||||||
|
#' @noRd
|
||||||
|
.require_balance_support <- function(con) {
|
||||||
|
if (.corpus_has_balance_subtype(con)) return(invisible(TRUE))
|
||||||
|
cli::cli_abort(
|
||||||
|
c("This corpus does not classify cash and security holdings.",
|
||||||
|
i = "`summary_categories` has no {.field balance_subtype} column.",
|
||||||
|
i = "Republish from cog_pipeline at #76/#77 or later."),
|
||||||
|
class = "uscogdata_no_balance_support"
|
||||||
|
)
|
||||||
|
}
|
||||||
@@ -0,0 +1,62 @@
|
|||||||
|
% Generated by roxygen2: do not edit by hand
|
||||||
|
% Please edit documentation in R/balances.R
|
||||||
|
\name{cog_balances}
|
||||||
|
\alias{cog_balances}
|
||||||
|
\title{Cash and security holdings for one or more governments}
|
||||||
|
\usage{
|
||||||
|
cog_balances(
|
||||||
|
govid,
|
||||||
|
years,
|
||||||
|
category = NULL,
|
||||||
|
per_capita = FALSE,
|
||||||
|
adjust_to_year = NULL,
|
||||||
|
basis = c("harmonized", "raw"),
|
||||||
|
recipe = NULL
|
||||||
|
)
|
||||||
|
}
|
||||||
|
\arguments{
|
||||||
|
\item{govid}{Canonical govid(s): a character vector, or a data frame with a
|
||||||
|
`canonical_govid` column (e.g. from [cog_gov_search()]).}
|
||||||
|
|
||||||
|
\item{years}{Integer vector of fiscal years.}
|
||||||
|
|
||||||
|
\item{category}{Optional character vector of categories to keep. One of
|
||||||
|
`"Fund Balances"`, `"Insurance Trust Balances"`,
|
||||||
|
`"Retirement System Holdings"`. There is deliberately no `subtype`
|
||||||
|
argument: for holdings, `category` is a strict coarsening of
|
||||||
|
`balance_subtype` (unlike the money verbs, where the two axes cross), so
|
||||||
|
every combination would be either redundant or empty.
|
||||||
|
`category = "Fund Balances"` is exactly the `general` family
|
||||||
|
(`W01`/`W31`/`W61`). `balance_subtype` is returned, so a finer split is
|
||||||
|
one `dplyr::filter()` away.}
|
||||||
|
|
||||||
|
\item{per_capita}{Divide holdings by population. Note this is a **stock per
|
||||||
|
resident** (reserves per person), which is *not* comparable to
|
||||||
|
[cog_spending()]'s per-capita figures -- those are a flow per person.}
|
||||||
|
|
||||||
|
\item{adjust_to_year}{Deflate to this year's dollars (CPI-U).}
|
||||||
|
|
||||||
|
\item{basis}{Accepted for uniformity with the money verbs, but currently a
|
||||||
|
**no-op**: `harmonization_map` carries no balance-code rows, so harmonized
|
||||||
|
and raw space are identical for holdings. Reported in
|
||||||
|
`provenance$basis_note`.}
|
||||||
|
|
||||||
|
\item{recipe}{Optional harmonization recipe id (see [cog_recipes()]).
|
||||||
|
`"cash_securities_z77_wide"` and `"cash_securities_z78_wide"` bridge the
|
||||||
|
wide era to the modern one.}
|
||||||
|
}
|
||||||
|
\value{
|
||||||
|
A `tbl_df` with a `provenance` attribute. Amounts are full US
|
||||||
|
dollars.
|
||||||
|
}
|
||||||
|
\description{
|
||||||
|
Returns Census cash-and-security holdings (`category_type = "balance"`):
|
||||||
|
fund balances, retirement system holdings and insurance trust balances.
|
||||||
|
}
|
||||||
|
\section{Holdings are not GAAP fund balance}{
|
||||||
|
|
||||||
|
Census holdings are **gross** -- no liabilities are netted -- so a reserve
|
||||||
|
ratio built from them overstates what is actually available. They are not
|
||||||
|
comparable to a GAAP fund balance from an ACFR.
|
||||||
|
}
|
||||||
|
|
||||||
@@ -97,3 +97,60 @@ test_that("balance views are skipped on a corpus without balance_subtype", {
|
|||||||
expect_true("revenue_long" %in% views)
|
expect_true("revenue_long" %in% views)
|
||||||
})
|
})
|
||||||
})
|
})
|
||||||
|
|
||||||
|
test_that("cog_balances returns holdings for a government that has them", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
r <- cog_balances("550000227544", 2019)
|
||||||
|
expect_s3_class(r, "tbl_df")
|
||||||
|
expect_true(nrow(r) > 0L)
|
||||||
|
expect_true(all(c("year", "canonical_govid", "gov_name", "balance_subtype",
|
||||||
|
"category", "amt_nominal") %in% names(r)))
|
||||||
|
expect_identical(sort(unique(r$category)),
|
||||||
|
c("Fund Balances", "Insurance Trust Balances"))
|
||||||
|
expect_false(is.null(attr(r, "provenance")))
|
||||||
|
expect_identical(attr(r, "provenance")$verb, "cog_balances")
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('category = "Fund Balances" is exactly the general family', {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
r <- cog_balances("550000227544", 2019, category = "Fund Balances")
|
||||||
|
expect_identical(unique(r$balance_subtype), "general")
|
||||||
|
codes <- sort(unlist(strsplit(paste(r$codes_included, collapse = ","), ",")))
|
||||||
|
expect_identical(codes, c("W01", "W31", "W61"))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("no flow code can reach cog_balances", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
r <- cog_balances("550000227544", c(2011, 2012, 2019, 2020))
|
||||||
|
got <- unique(unlist(strsplit(paste(r$codes_included, collapse = ","), ",")))
|
||||||
|
|
||||||
|
# The expected set is read from the RAW corpus, never from the verb --
|
||||||
|
# verifying an absence through the filter that creates it proves nothing.
|
||||||
|
ds <- arrow::open_dataset(file.path(fixture_corpus_path(), "data", "long"))
|
||||||
|
sc <- arrow::read_parquet(
|
||||||
|
file.path(fixture_corpus_path(), "data", "summary_categories.parquet"))
|
||||||
|
sc <- as.data.frame(sc)
|
||||||
|
balance_codes <- sc$item_code[sc$category_type == "balance"]
|
||||||
|
|
||||||
|
expect_true(all(got %in% balance_codes))
|
||||||
|
expect_true(length(setdiff(got, balance_codes)) == 0L)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("every balance_subtype maps to exactly one category", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
# Dropping the `subtype` argument is only safe while this tree holds. If the
|
||||||
|
# pipeline ever gives a balance subtype a second category, `category` becomes
|
||||||
|
# a lossy filter -- fail HERE rather than in a user's analysis.
|
||||||
|
sc <- as.data.frame(arrow::read_parquet(
|
||||||
|
file.path(fixture_corpus_path(), "data", "summary_categories.parquet")))
|
||||||
|
b <- sc[sc$category_type == "balance", ]
|
||||||
|
per_subtype <- tapply(b$category, b$balance_subtype,
|
||||||
|
function(x) length(unique(x)))
|
||||||
|
expect_true(all(per_subtype == 1L))
|
||||||
|
})
|
||||||
|
|||||||
Reference in New Issue
Block a user