feat: name a cohort by state/type predicate instead of a 40k-id IN list
cog_spending(), cog_revenue() and cog_balances() gain optional state/type
arguments. Both default to NULL, so every existing govid-based call is
unchanged.
The verbs took a cohort only as a govid vector, which .sql_lit_chr()
rendered into a quoted IN list and .verb_spendrev() embedded into 5-8
separate statements per call: the scope check, the main aggregate, the
per-capita join, the harmonization block, and the suggestion and
suppression queries. For type = "city" that list is 301,589 characters,
parsed and planned from scratch every time it appears.
Passing state/type instead expresses the cohort as a subquery against
canonical_fips_xwalk, so its size never enters the SQL string at all.
Measured on the production corpus, same FY2022 aggregate over the
20,106-government city cohort, DUCKDB_THREADS=2, median of 5:
IN (20,106 literals) -- 0.3.0 432 ms
join against a temp cohort table 132 ms
predicate on canonical_fips_xwalk 102 ms
no cohort filter at all (the floor) 105 ms
The predicate reaches the no-filter floor: the cohort restriction is
now free. End to end through cog_spending(category = "Police"),
1080 ms -> 271 ms, 3.99x -- larger than the single-query saving,
because the repetition across statements is what actually cost.
Design decisions, both made explicitly rather than left implicit:
- govid AND state/type INTERSECT. "These ids, narrowed to that
state/type" is a real query, and an error here could never be
relaxed later without breaking callers.
- A predicate cohort has no id list to report, so
provenance$scope$govids_found/govids_missing stay empty and a new
scope$cohort block carries state, type and n_governments. Resolving
the ids just to report them would put 20,000 govids in every
fleet-scale response body -- the cost this change removes. A
govid-named cohort's provenance is untouched.
state/type are coerced with .coerce_state_to_fips()/.coerce_type(), the
same helpers cog_gov_search() uses. That is load-bearing: the argument
is a postal abbreviation ("WI") while fips_state holds a FIPS code
("55"), and a predicate on the raw parameter matches nothing and returns
an empty result indistinguishable from "reported nothing". cog-api hit
exactly this trap optimizing the same path.
.attach_per_capita() now keys its population lookup on the govids present
in the result rather than the requested cohort. Those are the only ones
its LEFT JOIN can match, so the output is identical -- but it needs no id
list, and on a paginated call it looks up one page instead of the fleet.
Fixes uscogdata#58.
This commit is contained in:
+13
-7
@@ -18,7 +18,9 @@
|
||||
#' comparable to a GAAP fund balance from an ACFR.
|
||||
#'
|
||||
#' @param govid Canonical govid(s): a character vector, or a data frame with a
|
||||
#' `canonical_govid` column (e.g. from [cog_gov_search()]).
|
||||
#' `canonical_govid` column (e.g. from [cog_gov_search()]). `NULL` to name
|
||||
#' the cohort by `state`/`type` instead.
|
||||
#' @inheritParams cog_spending
|
||||
#' @param years Integer vector of fiscal years.
|
||||
#' @param category Optional character vector of categories to keep. One of
|
||||
#' `"Fund Balances"`, `"Insurance Trust Balances"`,
|
||||
@@ -63,15 +65,16 @@
|
||||
#' requested years). `expenditure_concept`/`revenue_concept` are `NA` --
|
||||
#' holdings are a stock, not a flow, so neither concept vocabulary applies.
|
||||
#' @export
|
||||
cog_balances <- function(govid, years, category = NULL,
|
||||
cog_balances <- function(govid = NULL, years, category = NULL,
|
||||
per_capita = FALSE, adjust_to_year = NULL,
|
||||
basis = c("harmonized", "raw"), recipe = NULL) {
|
||||
basis = c("harmonized", "raw"), recipe = NULL,
|
||||
state = NULL, type = NULL) {
|
||||
call <- match.call()
|
||||
basis <- match.arg(basis, c("harmonized", "raw"))
|
||||
# Coerce FIRST, validate second: .validate_verb_inputs() asserts
|
||||
# is.character(govid), and a data-frame govid (cog_gov_search() output) has
|
||||
# not been unwrapped yet at this point.
|
||||
govid <- .coerce_govid_input(govid)
|
||||
govid <- if (is.null(govid)) NULL else .coerce_govid_input(govid)
|
||||
# The money verbs' validator, reused rather than re-implemented (R/spending.R).
|
||||
# It covers the exact superset cog_balances() needs -- including the
|
||||
# recipe/category mutual-exclusivity guard -- so a second local copy would
|
||||
@@ -91,6 +94,8 @@ cog_balances <- function(govid, years, category = NULL,
|
||||
years <- as.integer(years)
|
||||
if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year)
|
||||
|
||||
cohort <- .make_cohort(govid, state, type)
|
||||
|
||||
con <- .ensure_session()
|
||||
.require_balance_support(con)
|
||||
scope <- .check_govids_in_scope(govid)
|
||||
@@ -109,7 +114,7 @@ cog_balances <- function(govid, years, category = NULL,
|
||||
.validate_recipe_id(con, recipe)
|
||||
comps <- .recipe_components(con, recipe)
|
||||
recipe_label <- comps$label[[1]]
|
||||
result <- .run_recipe(con, recipe, govid, years)
|
||||
result <- .run_recipe(con, recipe, cohort, years)
|
||||
sql <- attr(result, "sql_query")
|
||||
result <- .shape_recipe_result(result, "balance_subtype", recipe_label)
|
||||
recipe_block <- list(
|
||||
@@ -119,7 +124,7 @@ cog_balances <- function(govid, years, category = NULL,
|
||||
category_for_prov <- recipe_label
|
||||
} else {
|
||||
sql <- .build_verb_sql("balance_annotated", "balance_subtype",
|
||||
govid, years, category,
|
||||
cohort, years, category,
|
||||
ig_view = NULL, subtype_scope = NULL)
|
||||
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
}
|
||||
@@ -127,7 +132,7 @@ cog_balances <- function(govid, years, category = NULL,
|
||||
# Order matters (matches .verb_spendrev()): per-capita first, so
|
||||
# .attach_real_dollars() deflates the nominal per-capita column into
|
||||
# amt_per_capita_real rather than needing amt_per_capita_nominal recomputed.
|
||||
if (isTRUE(per_capita)) result <- .attach_per_capita(result, con, govid)
|
||||
if (isTRUE(per_capita)) result <- .attach_per_capita(result, con)
|
||||
if (!is.null(adjust_to_year)) {
|
||||
result <- .attach_real_dollars(result, adjust_to_year, per_capita)
|
||||
}
|
||||
@@ -145,6 +150,7 @@ cog_balances <- function(govid, years, category = NULL,
|
||||
)
|
||||
prov$scope$govids_found <- scope$found
|
||||
prov$scope$govids_missing <- scope$missing
|
||||
prov$scope$cohort <- .cohort_provenance(con, cohort)
|
||||
|
||||
prov$balance_caveats <- .balance_caveats(
|
||||
con, prov$codes_summed$observed, years
|
||||
|
||||
@@ -54,7 +54,7 @@
|
||||
#' expenditure_concept = "total": ig_long_harmonized COALESCEs rather than
|
||||
#' drops NULL-harmonized rows, so harmonization never excludes an IG row.
|
||||
#' @noRd
|
||||
.build_harmonization_block <- function(con, govid, years, resolved,
|
||||
.build_harmonization_block <- function(con, cohort, years, resolved,
|
||||
subtype_col, subtype_scope) {
|
||||
if (!identical(resolved$basis, "harmonized")) {
|
||||
return(list(
|
||||
@@ -68,12 +68,12 @@
|
||||
sql <- sprintf(
|
||||
"SELECT COUNT(*) AS n, COALESCE(SUM(amt), 0) * 1000.0 AS amt
|
||||
FROM long
|
||||
WHERE canonical_govid IN (%s) AND year IN (%s)
|
||||
WHERE %s AND year IN (%s)
|
||||
AND NOT is_aggregate AND harmonized_code IS NULL
|
||||
AND item_code IN (
|
||||
SELECT item_code FROM summary_categories WHERE %s IN (%s)
|
||||
)",
|
||||
.sql_lit_chr(govid), paste(as.integer(years), collapse = ","),
|
||||
.cohort_sql(cohort), paste(as.integer(years), collapse = ","),
|
||||
subtype_col, .sql_lit_chr(subtype_scope)
|
||||
)
|
||||
na <- DBI::dbGetQuery(con, sql)
|
||||
|
||||
+127
@@ -0,0 +1,127 @@
|
||||
# How a verb names the set of governments it queries.
|
||||
#
|
||||
# Historically there was one way: a `govid` character vector, rendered by
|
||||
# .sql_lit_chr() into a quoted IN list. That is fine for a handful of
|
||||
# governments and pathological for a fleet. Measured against the production
|
||||
# corpus, the same FY2022 aggregate over the 20,106-government `type = "city"`
|
||||
# cohort:
|
||||
#
|
||||
# cohort expressed as time
|
||||
# IN (20,106 literals) 449 ms
|
||||
# join against a temp cohort table 99 ms
|
||||
# predicate on canonical_fips_xwalk 94 ms
|
||||
# no cohort filter at all (the floor) 88 ms
|
||||
#
|
||||
# 4.8x, and within 7% of the no-filter floor. The rendered IN list is 301,591
|
||||
# characters and .verb_spendrev() embeds it in 5-8 separate statements per
|
||||
# call, so the parse-and-plan cost is paid over and over (uscogdata#58).
|
||||
#
|
||||
# A cohort therefore has two independent halves, and a query can carry either
|
||||
# or both:
|
||||
#
|
||||
# ids an explicit canonical_govid vector -> literal IN list
|
||||
# predicate state/type over canonical_fips_xwalk -> IN (SELECT ...)
|
||||
#
|
||||
# Both together is an INTERSECTION -- "these ids, narrowed to that state/type"
|
||||
# -- never a precedence rule where one silently wins.
|
||||
|
||||
#' Build the internal cohort object shared by every query verb.
|
||||
#'
|
||||
#' `state` and `type` are coerced with the SAME helpers `cog_gov_search()`
|
||||
#' uses. That is load-bearing, not tidiness: the public argument is a postal
|
||||
#' abbreviation (`"WI"`) while `canonical_fips_xwalk.fips_state` holds a FIPS
|
||||
#' code (`"55"`), and `type` is a label (`"city"`) against an integer
|
||||
#' `govs_type`. A predicate written against the raw parameter matches nothing
|
||||
#' and returns an empty result indistinguishable from "this government
|
||||
#' reported nothing" -- cog-api hit exactly that trap optimizing this path.
|
||||
#' One definition of the translation, not two.
|
||||
#'
|
||||
#' @param govid Already-coerced character vector of canonical_govids, or NULL.
|
||||
#' @param state Postal abbreviation or FIPS code, or NULL.
|
||||
#' @param type Type label or integer code, or NULL.
|
||||
#' @noRd
|
||||
.make_cohort <- function(govid = NULL, state = NULL, type = NULL) {
|
||||
if (is.null(govid) && is.null(state) && is.null(type)) {
|
||||
cli::cli_abort(c(
|
||||
"A cohort must be named.",
|
||||
"*" = "Pass {.arg govid} for specific governments, or {.arg state}/{.arg type} for every government matching a predicate.",
|
||||
"i" = "Passing both intersects them: the governments in {.arg govid} that also match {.arg state}/{.arg type}."
|
||||
), class = "uscogdata_no_cohort")
|
||||
}
|
||||
structure(
|
||||
list(
|
||||
ids = govid,
|
||||
state = state,
|
||||
type = type,
|
||||
state_fips = if (is.null(state)) NULL else .coerce_state_to_fips(state),
|
||||
type_int = if (is.null(type)) NULL else .coerce_type(type)
|
||||
),
|
||||
class = "uscogdata_cohort"
|
||||
)
|
||||
}
|
||||
|
||||
#' Is any part of this cohort expressed as an xwalk predicate?
|
||||
#' @noRd
|
||||
.cohort_by_predicate <- function(cohort) {
|
||||
!is.null(cohort$state_fips) || !is.null(cohort$type_int)
|
||||
}
|
||||
|
||||
#' Render the cohort as a SQL boolean expression over `col`.
|
||||
#'
|
||||
#' `col` may be qualified (`"l.canonical_govid"`, `"x.canonical_govid"`) --
|
||||
#' several call sites join the xwalk under an alias. The subquery's own
|
||||
#' projected column stays unqualified: it selects from canonical_fips_xwalk,
|
||||
#' not from the outer relation.
|
||||
#' @noRd
|
||||
.cohort_sql <- function(cohort, col = "canonical_govid") {
|
||||
preds <- character(0)
|
||||
|
||||
if (!is.null(cohort$ids)) {
|
||||
preds <- c(preds, sprintf("%s IN (%s)", col, .sql_lit_chr(cohort$ids)))
|
||||
}
|
||||
|
||||
if (.cohort_by_predicate(cohort)) {
|
||||
xwalk_preds <- character(0)
|
||||
if (!is.null(cohort$state_fips)) {
|
||||
xwalk_preds <- c(xwalk_preds,
|
||||
sprintf("fips_state = %s", .sql_lit_chr(cohort$state_fips)))
|
||||
}
|
||||
if (!is.null(cohort$type_int)) {
|
||||
xwalk_preds <- c(xwalk_preds, sprintf("govs_type = %d", cohort$type_int))
|
||||
}
|
||||
preds <- c(preds, sprintf(
|
||||
"%s IN (SELECT canonical_govid FROM canonical_fips_xwalk WHERE %s)",
|
||||
col, paste(xwalk_preds, collapse = " AND ")
|
||||
))
|
||||
}
|
||||
|
||||
paste(preds, collapse = " AND ")
|
||||
}
|
||||
|
||||
#' How many governments the cohort covers.
|
||||
#'
|
||||
#' One COUNT against the crosswalk, used only to populate the provenance
|
||||
#' `scope$cohort` block. Deliberately a count rather than the id list: a
|
||||
#' fleet-scale cohort would otherwise put 20,000 ids into every response body,
|
||||
#' which is the cost this issue exists to remove.
|
||||
#' @noRd
|
||||
.cohort_count <- function(con, cohort) {
|
||||
sql <- sprintf(
|
||||
"SELECT COUNT(*) AS n FROM canonical_fips_xwalk WHERE %s",
|
||||
.cohort_sql(cohort)
|
||||
)
|
||||
as.integer(DBI::dbGetQuery(con, sql)$n[[1]])
|
||||
}
|
||||
|
||||
#' The provenance `scope$cohort` block for a predicate cohort, or NULL when
|
||||
#' the cohort was named by id alone (in which case `govids_found`/
|
||||
#' `govids_missing` already describe it exactly).
|
||||
#' @noRd
|
||||
.cohort_provenance <- function(con, cohort) {
|
||||
if (!.cohort_by_predicate(cohort)) return(NULL)
|
||||
list(
|
||||
state = if (is.null(cohort$state)) NA_character_ else as.character(cohort$state),
|
||||
type = if (is.null(cohort$type)) NA_character_ else as.character(cohort$type),
|
||||
n_governments = .cohort_count(con, cohort)
|
||||
)
|
||||
}
|
||||
+5
-5
@@ -57,7 +57,7 @@
|
||||
#' aggregate rows. Without it the grid would offer cells the verb structurally
|
||||
#' never returns, so every one of them would fill as a phantom $0.
|
||||
#' @noRd
|
||||
.completion_grid_sql <- function(subtype_col, govid, years, category,
|
||||
.completion_grid_sql <- function(subtype_col, cohort, years, category,
|
||||
subtype_scope) {
|
||||
category_pred <- if (is.null(category)) {
|
||||
""
|
||||
@@ -76,13 +76,13 @@
|
||||
JOIN canonical_fips_xwalk x ON x.govs_type = cs.type
|
||||
JOIN summary_categories c ON c.item_code = cs.item_code
|
||||
JOIN representation r ON r.year = cs.year
|
||||
WHERE x.canonical_govid IN (%2$s)
|
||||
WHERE %2$s
|
||||
AND cs.year IN (%3$s)
|
||||
AND NOT cs.is_aggregate
|
||||
AND c.category IS NOT NULL
|
||||
AND c.%1$s IN (%4$s)
|
||||
%5$s",
|
||||
subtype_col, .sql_lit_chr(govid),
|
||||
subtype_col, .cohort_sql(cohort, "x.canonical_govid"),
|
||||
paste(as.integer(years), collapse = ","),
|
||||
.sql_lit_chr(subtype_scope), category_pred
|
||||
)
|
||||
@@ -94,10 +94,10 @@
|
||||
#' provenance block. Reported rows are passed through untouched -- filling
|
||||
#' must never alter or drop what the corpus actually published.
|
||||
#' @noRd
|
||||
.complete_result <- function(result, con, subtype_col, govid, years, category,
|
||||
.complete_result <- function(result, con, subtype_col, cohort, years, category,
|
||||
subtype_scope) {
|
||||
grid <- tibble::as_tibble(DBI::dbGetQuery(
|
||||
con, .completion_grid_sql(subtype_col, govid, years, category, subtype_scope)
|
||||
con, .completion_grid_sql(subtype_col, cohort, years, category, subtype_scope)
|
||||
))
|
||||
|
||||
result$value_source <- rep("reported", nrow(result))
|
||||
|
||||
+3
-3
@@ -113,7 +113,7 @@ cog_recipes <- function(pattern = NULL) {
|
||||
#' `year_min`/`year_max` -- so there is no double-counting. (Checkpoint
|
||||
#' review docs/phase_r_harmonization_review.md § 0.2.)
|
||||
#' @noRd
|
||||
.run_recipe <- function(con, recipe_id, govid, years) {
|
||||
.run_recipe <- function(con, recipe_id, cohort, years) {
|
||||
sql <- sprintf(
|
||||
"SELECT l.year, l.canonical_govid,
|
||||
COALESCE(x.gov_name, l.gov_name) AS gov_name,
|
||||
@@ -128,11 +128,11 @@ cog_recipes <- function(pattern = NULL) {
|
||||
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
|
||||
WHERE r.recipe_id = %1$s
|
||||
AND l.canonical_govid IN (%2$s)
|
||||
AND %2$s
|
||||
AND l.year IN (%3$s)
|
||||
GROUP BY 1, 2, 3
|
||||
ORDER BY 1, 2",
|
||||
.sql_lit_chr(recipe_id), .sql_lit_chr(govid),
|
||||
.sql_lit_chr(recipe_id), .cohort_sql(cohort, "l.canonical_govid"),
|
||||
paste(as.integer(years), collapse = ",")
|
||||
)
|
||||
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
|
||||
+6
-3
@@ -50,11 +50,12 @@
|
||||
#' optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`,
|
||||
#' and `value_source` when `complete = TRUE`.
|
||||
#' @export
|
||||
cog_revenue <- function(govid, years, category = NULL,
|
||||
cog_revenue <- function(govid = NULL, years, category = NULL,
|
||||
per_capita = FALSE, adjust_to_year = NULL,
|
||||
basis = c("harmonized", "raw"), recipe = NULL,
|
||||
revenue_concept = c("general", "total"),
|
||||
complete = FALSE, limit = NULL, offset = NULL) {
|
||||
complete = FALSE, limit = NULL, offset = NULL,
|
||||
state = NULL, type = NULL) {
|
||||
# flow_prefixes no longer classifies rows (crosswalk revenue_subtype
|
||||
# membership does -- General Revenue, i.e. everything except
|
||||
# insurance_trust) -- it only scopes the recipe-suggestion machinery to
|
||||
@@ -75,6 +76,8 @@ cog_revenue <- function(govid, years, category = NULL,
|
||||
revenue_concept = revenue_concept,
|
||||
complete = complete,
|
||||
limit = limit,
|
||||
offset = offset
|
||||
offset = offset,
|
||||
state = state,
|
||||
type = type
|
||||
)
|
||||
}
|
||||
|
||||
+11
-4
@@ -184,11 +184,18 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) {
|
||||
if (!is.character(state) || length(state) != 1L) {
|
||||
cli::cli_abort("`state` must be a 2-letter USPS abbrev or a FIPS integer.")
|
||||
}
|
||||
fips <- .state_abbrev_to_fips[[toupper(state)]]
|
||||
if (is.null(fips)) {
|
||||
cli::cli_abort("Unknown state abbreviation: {state}.")
|
||||
# Membership tested before the lookup, not after: `.state_abbrev_to_fips` is
|
||||
# a named CHARACTER vector, and `[[` on a name it does not carry throws
|
||||
# base R's "subscript out of bounds" rather than returning NULL -- which
|
||||
# made the curated message below unreachable dead code. Reported as a bare
|
||||
# subscript error, `cog_gov_search(state = "ZZ")` gave no hint that the
|
||||
# argument wants a postal abbreviation.
|
||||
key <- toupper(state)
|
||||
if (!key %in% names(.state_abbrev_to_fips)) {
|
||||
cli::cli_abort("Unknown state abbreviation: {state}.",
|
||||
class = "uscogdata_unknown_state")
|
||||
}
|
||||
fips
|
||||
.state_abbrev_to_fips[[key]]
|
||||
}
|
||||
|
||||
# USPS state / territory abbreviation -> 2-digit FIPS code.
|
||||
|
||||
+74
-20
@@ -68,7 +68,9 @@
|
||||
#' millions/billions). The conversion is recorded in the provenance attribute
|
||||
#' under `transformations$units_conversion`.
|
||||
#'
|
||||
#' @param govid Character vector of `canonical_govid` values.
|
||||
#' @param govid Character vector of `canonical_govid` values, or `NULL` to name
|
||||
#' the cohort by `state`/`type` instead. One of `govid`, `state`, or `type`
|
||||
#' is required.
|
||||
#' @param years Integer vector of years.
|
||||
#' @param category Character vector of category names (from
|
||||
#' `summary_categories.category`), or `NULL` for all categories broken out
|
||||
@@ -185,6 +187,29 @@
|
||||
#' `recipe` and with `complete = TRUE` -- see `offset` and `total_rows`.
|
||||
#' @param offset Rows to skip before `limit` starts counting (0-based).
|
||||
#' Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set.
|
||||
#' @param state,type Name the cohort by predicate instead of by id: `state` is
|
||||
#' a 2-letter USPS abbreviation (or a FIPS code) and `type` is one of
|
||||
#' `"state"`, `"county"`, `"city"`, `"township"` (or the integer `0:3`) --
|
||||
#' the same vocabulary, and the same internal coercion, as
|
||||
#' [cog_gov_search()]. Both default to `NULL`.
|
||||
#'
|
||||
#' The cohort is then expressed as a subquery against `canonical_fips_xwalk`
|
||||
#' inside each statement rather than round-tripped through R as a literal id
|
||||
#' list. For a fleet-scale cohort that is the difference between a
|
||||
#' 301,591-character `IN` list re-parsed in 5--8 statements per call and a
|
||||
#' constant-size predicate: measured at **94 ms versus 449 ms** for the same
|
||||
#' FY2022 aggregate over the 20,106-government `type = "city"` cohort, within
|
||||
#' 7% of the no-filter floor.
|
||||
#'
|
||||
#' Supplying `govid` **and** `state`/`type` INTERSECTS them -- the
|
||||
#' governments in `govid` that also match the predicate -- rather than one
|
||||
#' silently taking precedence. Naming no cohort at all (`govid`, `state` and
|
||||
#' `type` all `NULL`) aborts with class `uscogdata_no_cohort`.
|
||||
#'
|
||||
#' When the cohort is named by predicate, `provenance$scope$govids_found`
|
||||
#' and `govids_missing` are empty -- there is no id list to report against --
|
||||
#' and `provenance$scope$cohort` carries `state`, `type` and
|
||||
#' `n_governments` instead. A `govid`-named cohort reports exactly as before.
|
||||
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||
#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
||||
#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
||||
@@ -198,11 +223,12 @@
|
||||
#' round trip -- so a caller walking pages never has to ask "how many are
|
||||
#' there" separately.
|
||||
#' @export
|
||||
cog_spending <- function(govid, years, category = NULL,
|
||||
cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
per_capita = FALSE, adjust_to_year = NULL,
|
||||
basis = c("harmonized", "raw"), recipe = NULL,
|
||||
expenditure_concept = c("primary", "direct", "total"),
|
||||
complete = FALSE, limit = NULL, offset = NULL) {
|
||||
complete = FALSE, limit = NULL, offset = NULL,
|
||||
state = NULL, type = NULL) {
|
||||
# flow_prefixes no longer classifies rows (crosswalk subtype membership
|
||||
# does, per expenditure_concept) -- it only scopes the recipe-suggestion
|
||||
# machinery to this verb's recipe families (see R/suggestions.R; the
|
||||
@@ -223,7 +249,9 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
expenditure_concept = expenditure_concept,
|
||||
complete = complete,
|
||||
limit = limit,
|
||||
offset = offset
|
||||
offset = offset,
|
||||
state = state,
|
||||
type = type
|
||||
)
|
||||
}
|
||||
|
||||
@@ -251,7 +279,8 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
basis = c("harmonized", "raw"), recipe = NULL,
|
||||
expenditure_concept = c("primary", "direct", "total"),
|
||||
revenue_concept = c("general", "total"),
|
||||
complete = FALSE, limit = NULL, offset = NULL) {
|
||||
complete = FALSE, limit = NULL, offset = NULL,
|
||||
state = NULL, type = NULL) {
|
||||
basis_explicit <- length(basis) == 1L
|
||||
basis <- match.arg(basis, c("harmonized", "raw"))
|
||||
# match.arg() itself throws a base `simpleError`, not an rlang-classed
|
||||
@@ -291,7 +320,10 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
.revenue_concept_subtypes(revenue_concept)
|
||||
}
|
||||
|
||||
govid <- .coerce_govid_input(govid, arg = "govid")
|
||||
# NULL `govid` means "the cohort is named by predicate"; anything else is
|
||||
# coerced and validated exactly as before, so an empty or wrong-typed vector
|
||||
# still fails with its original message rather than being read as absent.
|
||||
govid <- if (is.null(govid)) NULL else .coerce_govid_input(govid, arg = "govid")
|
||||
# allow_all_categories = TRUE: cog_spending()/cog_revenue() are the two
|
||||
# verbs the reserved pseudo-category is defined for. cog_balances() shares
|
||||
# this validator but leaves the argument at its FALSE default, so it
|
||||
@@ -300,6 +332,10 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
.validate_verb_inputs(govid, years, category, per_capita, adjust_to_year,
|
||||
recipe, allow_all_categories = TRUE)
|
||||
|
||||
# Built after validation so the argument-shape errors above keep firing
|
||||
# first, and before .ensure_session() so a bad state/type costs no I/O.
|
||||
cohort <- .make_cohort(govid, state, type)
|
||||
|
||||
# Recognize the reserved pseudo-category. Detected after type validation so a
|
||||
# non-character `category` still fails with the ordinary type error.
|
||||
all_categories <- !is.null(category) && .ALL_CATEGORIES %in% category
|
||||
@@ -407,7 +443,7 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
.validate_recipe_id(con, recipe)
|
||||
comps <- .recipe_components(con, recipe)
|
||||
recipe_label <- comps$label[[1]]
|
||||
result <- .run_recipe(con, recipe, govid, years)
|
||||
result <- .run_recipe(con, recipe, cohort, years)
|
||||
sql <- attr(result, "sql_query")
|
||||
result <- .shape_recipe_result(result, subtype_col, recipe_label)
|
||||
recipe_block <- list(
|
||||
@@ -423,7 +459,7 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
} else {
|
||||
NULL
|
||||
}
|
||||
sql <- .build_verb_sql(view, subtype_col, govid, years,
|
||||
sql <- .build_verb_sql(view, subtype_col, cohort, years,
|
||||
if (all_categories) NULL else category,
|
||||
ig_view, subtype_scope,
|
||||
all_categories = all_categories,
|
||||
@@ -441,7 +477,7 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
} else {
|
||||
count_sql <- sprintf(
|
||||
"SELECT COUNT(*) AS n FROM (%s) AS _uncounted",
|
||||
.build_verb_sql(view, subtype_col, govid, years,
|
||||
.build_verb_sql(view, subtype_col, cohort, years,
|
||||
if (all_categories) NULL else category,
|
||||
ig_view, subtype_scope,
|
||||
all_categories = all_categories)
|
||||
@@ -457,13 +493,13 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
# spurious 0.
|
||||
completion <- list(applied = FALSE, rows_filled = 0L, absence_means = list())
|
||||
if (complete) {
|
||||
result <- .complete_result(result, con, subtype_col, govid, years,
|
||||
result <- .complete_result(result, con, subtype_col, cohort, years,
|
||||
category, subtype_scope)
|
||||
completion <- attr(result, ".completion")
|
||||
attr(result, ".completion") <- NULL
|
||||
}
|
||||
|
||||
if (per_capita) result <- .attach_per_capita(result, con, govid)
|
||||
if (per_capita) result <- .attach_per_capita(result, con)
|
||||
if (!is.null(adjust_to_year)) {
|
||||
result <- .attach_real_dollars(result, adjust_to_year, per_capita)
|
||||
}
|
||||
@@ -491,7 +527,7 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
basis_for_prov <- resolved$basis
|
||||
basis_note_for_prov <- resolved$note
|
||||
harmonization <- .build_harmonization_block(
|
||||
con, govid, years, resolved, subtype_col, subtype_scope
|
||||
con, cohort, years, resolved, subtype_col, subtype_scope
|
||||
)
|
||||
# C1(a): gap detection must run against the Direct leg alone. `result`
|
||||
# can also carry UNION'd intergovernmental rows (expenditure_concept =
|
||||
@@ -506,7 +542,7 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
} else {
|
||||
result
|
||||
}
|
||||
suggestions <- .build_suggestions(con, govid, years, category,
|
||||
suggestions <- .build_suggestions(con, cohort, years, category,
|
||||
direct_leg_result,
|
||||
resolved$basis, flow_prefixes,
|
||||
.select_long_view(view_base, resolved$basis),
|
||||
@@ -611,6 +647,12 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
)
|
||||
prov$scope$govids_found <- scope$found
|
||||
prov$scope$govids_missing <- scope$missing
|
||||
# A predicate-named cohort has no id list to report found/missing against
|
||||
# (both stay empty), so it describes itself instead. Deliberately a COUNT
|
||||
# rather than the resolved ids: enumerating them would put 20,000 govids in
|
||||
# every fleet-scale response body, which is the cost this path exists to
|
||||
# remove. NULL for a govid-named cohort, so that output is untouched.
|
||||
prov$scope$cohort <- .cohort_provenance(con, cohort)
|
||||
attr(result, "provenance") <- prov
|
||||
attr(result, ".popyear_range") <- NULL
|
||||
# Attached here, after every downstream transform (per_capita/real-dollar
|
||||
@@ -639,7 +681,11 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
.validate_verb_inputs <- function(govid, years, category,
|
||||
per_capita, adjust_to_year, recipe = NULL,
|
||||
allow_all_categories = FALSE) {
|
||||
if (!is.character(govid) || length(govid) == 0L) {
|
||||
# NULL is allowed only because the caller has already established that the
|
||||
# cohort is named some other way (`state`/`type`); .make_cohort() is what
|
||||
# refuses a call that names no cohort at all. A supplied-but-empty `govid`
|
||||
# still fails here, exactly as before.
|
||||
if (!is.null(govid) && (!is.character(govid) || length(govid) == 0L)) {
|
||||
cli::cli_abort("`govid` must be a non-empty character vector.")
|
||||
}
|
||||
if (!(is.integer(years) || is.numeric(years)) || length(years) == 0L) {
|
||||
@@ -740,10 +786,10 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
}
|
||||
|
||||
#' @noRd
|
||||
.build_verb_sql <- function(view, subtype_col, govid, years, category,
|
||||
.build_verb_sql <- function(view, subtype_col, cohort, years, category,
|
||||
ig_view = NULL, subtype_scope = NULL,
|
||||
all_categories = FALSE, limit = NULL, offset = NULL) {
|
||||
govid_lit <- .sql_lit_chr(govid)
|
||||
cohort_pred <- .cohort_sql(cohort)
|
||||
years_lit <- paste(as.integer(years), collapse = ",")
|
||||
# In all-categories mode there is no category filter: the sum is defined by
|
||||
# the concept's SUBTYPE allowlist (subtype_pred below), which is the real
|
||||
@@ -813,13 +859,13 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS codes_included,
|
||||
bool_or(is_aggregate) AS aggregate_fallback
|
||||
FROM %2$s
|
||||
WHERE canonical_govid IN (%3$s)
|
||||
WHERE %3$s
|
||||
AND year IN (%4$s)
|
||||
%5$s
|
||||
%6$s
|
||||
GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, %1$s%8$s
|
||||
ORDER BY year, canonical_govid, %1$s%8$s",
|
||||
subtype_col, source_expr, govid_lit, years_lit, category_pred, subtype_pred,
|
||||
subtype_col, source_expr, cohort_pred, years_lit, category_pred, subtype_pred,
|
||||
category_select, category_group
|
||||
)
|
||||
|
||||
@@ -845,8 +891,16 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
}
|
||||
}
|
||||
|
||||
#' Join population onto a result and derive the per-capita columns.
|
||||
#'
|
||||
#' The population lookup is keyed on the govids PRESENT IN `result`, not on the
|
||||
#' cohort that produced it. Those are the only ones the LEFT JOIN below can
|
||||
#' match, so the joined output is identical either way -- but it means this
|
||||
#' works unchanged for a cohort named by predicate (where no id list exists in
|
||||
#' R at all), and on a paginated call it looks up one page's governments
|
||||
#' instead of the whole fleet's.
|
||||
#' @noRd
|
||||
.attach_per_capita <- function(result, con, govid) {
|
||||
.attach_per_capita <- function(result, con) {
|
||||
if (nrow(result) == 0L) {
|
||||
result$amt_per_capita_nominal <- numeric(0)
|
||||
result$pop_source <- character(0)
|
||||
@@ -859,7 +913,7 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
FROM gov_population_yearly
|
||||
WHERE canonical_govid IN (%s)
|
||||
AND year IN (%s)",
|
||||
.sql_lit_chr(govid), years_lit
|
||||
.sql_lit_chr(unique(result$canonical_govid)), years_lit
|
||||
)
|
||||
pops <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
result <- dplyr::left_join(result, pops,
|
||||
|
||||
+6
-6
@@ -42,8 +42,8 @@
|
||||
#' verb call: recipes whose generic join would fill a real gap in `result`.
|
||||
#'
|
||||
#' @param con Active DuckDB connection.
|
||||
#' @param govid Character vector of canonical_govid values (the verb's raw
|
||||
#' `govid`).
|
||||
#' @param cohort The verb's cohort object (see `.make_cohort()`), naming the
|
||||
#' governments by id, by state/type predicate, or both.
|
||||
#' @param years Integer vector of requested years.
|
||||
#' @param category `category` argument as passed to the verb (character
|
||||
#' vector or `NULL`; suggestions are only computed when non-NULL).
|
||||
@@ -85,7 +85,7 @@
|
||||
#' ig_recipe_id, trigger, suppressed_amount, suppressed_years,
|
||||
#' suppressed_codes)`, possibly empty.
|
||||
#' @noRd
|
||||
.build_suggestions <- function(con, govid, years, category, result, basis,
|
||||
.build_suggestions <- function(con, cohort, years, category, result, basis,
|
||||
flow_prefixes, long_view,
|
||||
all_categories = FALSE,
|
||||
subtype_col = NULL, subtype_scope = NULL) {
|
||||
@@ -160,7 +160,7 @@
|
||||
# violation would kill signposting, the exact failure class uscogdata#9
|
||||
# exists to prevent. Owner's call: keep this simple; a batch-aware
|
||||
# optimization, if one is worth building, is a separate issue.
|
||||
supp <- .suppressed_components(con, candidates, govid, years, long_view, flow_prefixes)
|
||||
supp <- .suppressed_components(con, candidates, cohort, years, long_view, flow_prefixes)
|
||||
|
||||
if (length(gap_years) == 0L && nrow(supp) == 0L) return(list())
|
||||
|
||||
@@ -188,9 +188,9 @@
|
||||
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||
WHERE r.recipe_id IN (%s)
|
||||
AND l.canonical_govid IN (%s)
|
||||
AND %s
|
||||
AND l.year IN (%s)",
|
||||
.sql_lit_chr(candidates), .sql_lit_chr(govid),
|
||||
.sql_lit_chr(candidates), .cohort_sql(cohort, "l.canonical_govid"),
|
||||
paste(gap_years, collapse = ",")
|
||||
))
|
||||
}
|
||||
|
||||
+8
-6
@@ -50,7 +50,9 @@
|
||||
#'
|
||||
#' @param con Active DuckDB connection.
|
||||
#' @param candidates Character vector of recipe ids to measure.
|
||||
#' @param govid Character vector of canonical_govid values.
|
||||
#' @param cohort The verb's cohort object (see `.make_cohort()`), rendered
|
||||
#' into the govid predicate on both the outer scan and the restated
|
||||
#' NOT EXISTS filter.
|
||||
#' @param years Integer vector of requested years.
|
||||
#' @param long_view Name of the verb's long view, from `.select_long_view()`.
|
||||
#' @param flow_prefixes The calling verb's own flow-type prefixes (see
|
||||
@@ -60,7 +62,7 @@
|
||||
#' dollars), `suppressed_codes` (comma-joined, sorted). Zero rows when
|
||||
#' nothing is suppressed.
|
||||
#' @noRd
|
||||
.suppressed_components <- function(con, candidates, govid, years, long_view,
|
||||
.suppressed_components <- function(con, candidates, cohort, years, long_view,
|
||||
flow_prefixes) {
|
||||
empty <- tibble::tibble(
|
||||
recipe_id = character(0), year = numeric(0),
|
||||
@@ -93,7 +95,7 @@
|
||||
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||
WHERE r.recipe_id IN (%1$s)
|
||||
AND l.canonical_govid IN (%2$s)
|
||||
AND %2$s
|
||||
AND l.year IN (%3$s)
|
||||
AND l.amt <> 0
|
||||
AND LEFT(r.component_code, 1) IN (%5$s)
|
||||
@@ -103,13 +105,13 @@
|
||||
AND v.year = l.year
|
||||
AND v.item_code = l.item_code
|
||||
AND v.year IN (%3$s) -- restated: enables partition pruning (I3a)
|
||||
AND v.canonical_govid IN (%2$s) -- restated: pushes the govid filter (I3a)
|
||||
AND %6$s -- restated: pushes the cohort filter (I3a)
|
||||
)
|
||||
GROUP BY 1, 2
|
||||
ORDER BY 1, 2",
|
||||
.sql_lit_chr(candidates), .sql_lit_chr(govid),
|
||||
.sql_lit_chr(candidates), .cohort_sql(cohort, "l.canonical_govid"),
|
||||
paste(as.integer(years), collapse = ","), long_view,
|
||||
.sql_lit_chr(flow_prefixes)
|
||||
.sql_lit_chr(flow_prefixes), .cohort_sql(cohort, "v.canonical_govid")
|
||||
)
|
||||
tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user