Files
uscogdata/R/session.R
jared 22c2478634 fix(balances): validate the full signature, surface caveats in cog_explain, memoise coverage windows (#25)
Final-review findings F-1, F-2, F-6, F-8 (plus the F-9 @return reword,
which shares R/balances.R).

F-2: .validate_balance_inputs() checked 2 of cog_balances()' 7 arguments.
years = integer(0) leaked a raw DuckDB 'Parser Error ... AND year IN ()'
with the generated SQL echoed back; govid = character(0) and a non-character
category returned 0 rows with no error at all; recipe = c("a","b") threw
'the condition has length > 1' from inside .validate_recipe_id(). Replaced
with a call to the money verbs' own .validate_verb_inputs() (R/spending.R),
which validates the exact superset needed. Deleted the local copy rather
than extending it -- two validators is how they drift. Placed AFTER
.coerce_govid_input(), because .validate_verb_inputs() asserts
is.character(govid) and a data-frame govid is not unwrapped before that.
This is helper reuse of the same kind as .build_verb_sql()/.attach_per_capita();
the verb still does NOT route through .verb_spendrev().

F-1: falls out of F-2 for free -- the recipe/category mutual-exclusivity
guard lives inside .validate_verb_inputs(). Previously recipe silently
discarded category AND overwrote provenance$category with the recipe label,
so a caller asking for Fund Balances got X40/Z77 insurance-trust holdings
with no trace of the dropped filter.

F-6: cog_explain() rendered every provenance caveat block except
balance_caveats. Since .emit_balance_caveats() fires at most once per
session -- and is routinely consumed by a suppressMessages() call or an
unread knitr chunk -- cog_explain() is the only surface left for a caller
who deliberately audits the result. Added a 'Holdings caveats' section
guarded on !is.null(prov$balance_caveats). Also relabels the cosmetic
'Concept: NA' line on balance results as 'not applicable (holdings are a
stock, not a flow)'.

F-8: the coverage-window query has no govid and no year predicate -- its
answer depends only on the mounted corpus -- yet it scanned all of
balance_long on every call (35% of verb runtime on the fixture, and a
per-request throughput ceiling for cog-api#26). Memoised in
.uscogdata_env$balance_coverage_windows, invalidated by cog_close(), the
same pattern as .uscogdata_env$manifest.
2026-08-03 11:36:18 -04:00

102 lines
3.5 KiB
R

# R/session.R
#' Internal: open session, register views, cache manifest.
#' Not exported. Called lazily by verbs via .ensure_session().
#' @noRd
cog_open <- function(url = .resolve_url(),
cache_dir = .resolve_cache_dir()) {
.check_url_configured(url)
if (!dir.exists(cache_dir)) dir.create(cache_dir, recursive = TRUE)
con <- DBI::dbConnect(duckdb::duckdb())
DBI::dbExecute(con, "INSTALL httpfs; LOAD httpfs;")
manifest <- .fetch_or_cache_manifest(url, cache_dir)
.validate_schema(manifest, supported = c(4L, 5L, 6L))
.validate_scope(manifest)
.register_views(con, url, manifest)
.uscogdata_env$con <- con
.uscogdata_env$manifest <- manifest
.uscogdata_env$url <- url
.uscogdata_env$cache_dir <- cache_dir
invisible(con)
}
#' @noRd
.ensure_session <- function() {
if (is.null(.uscogdata_env$con) ||
!DBI::dbIsValid(.uscogdata_env$con)) {
cog_open()
}
.uscogdata_env$con
}
# Coerce an input to a character vector of canonical_govid values.
# Accepts either a character vector (returned as-is after `as.character`)
# or a data.frame / tibble with a `canonical_govid` column (such as the
# output of cog_gov_search() or cog_find_peers()) — in that case the
# column is extracted so results from discovery verbs can pipe directly
# into the query verbs.
#' @noRd
.coerce_govid_input <- function(x, arg = "govid") {
if (is.data.frame(x)) {
if (!"canonical_govid" %in% names(x)) {
cli::cli_abort(c(
"`{arg}` data frame must have a `canonical_govid` column.",
i = "Use the result of cog_gov_search() or cog_find_peers() directly, or pass a character vector of canonical_govids."
))
}
return(as.character(x$canonical_govid))
}
if (!is.character(x) && !is.numeric(x)) {
cli::cli_abort(
"`{arg}` must be a character vector or a data frame with a `canonical_govid` column."
)
}
as.character(x)
}
# Check which of the supplied govids exist in canonical_fips_xwalk.
# Emits a cli message listing any missing ones alongside a pointer to the
# v0.1 scope explanation; returns both sets so callers can attach them to
# provenance.
#' @noRd
.check_govids_in_scope <- function(govids) {
govids <- unique(as.character(govids))
if (length(govids) == 0L) return(list(found = character(0), missing = character(0)))
con <- .ensure_session()
sql <- sprintf(
"SELECT canonical_govid FROM canonical_fips_xwalk WHERE canonical_govid IN (%s)",
.sql_lit_chr(govids)
)
found <- DBI::dbGetQuery(con, sql)$canonical_govid
missing <- setdiff(govids, found)
if (length(missing) > 0L) {
n <- length(missing)
shown <- paste(utils::head(missing, 5L), collapse = ", ")
more <- if (n > 5L) sprintf(" (+%d more)", n - 5L) else ""
cli::cli_inform(c(
i = sprintf("%d govid%s not found in v0.1 corpus: %s%s",
n, if (n == 1L) "" else "s", shown, more),
i = "Common causes: typo, pre-2017 PID that isn't bridged, or a scope-excluded type (4=special district, 5=school district).",
i = "Resolve canonical names with cog_gov_search() first."
))
}
list(found = found, missing = missing)
}
#' @noRd
cog_close <- function() {
if (!is.null(.uscogdata_env$con) && DBI::dbIsValid(.uscogdata_env$con)) {
DBI::dbDisconnect(.uscogdata_env$con, shutdown = TRUE)
}
.uscogdata_env$con <- NULL
.uscogdata_env$manifest <- NULL
.uscogdata_env$balance_caveats_shown <- NULL
# Memoised corpus-constant; a different corpus may be mounted next.
.uscogdata_env$balance_coverage_windows <- NULL
}