diff --git a/DESCRIPTION b/DESCRIPTION index 7159bab..ec395ac 100644 --- a/DESCRIPTION +++ b/DESCRIPTION @@ -1,7 +1,7 @@ Package: uscogdata Type: Package Title: Curated Reader for the Civilytics US Census of Governments Finance Corpus -Version: 0.3.0 +Version: 0.4.0 Authors@R: c( person(c("Jared", "E."), "Knowles", email = "jared@civilytics.com", diff --git a/NEWS.md b/NEWS.md index cace5cd..6c3b108 100644 --- a/NEWS.md +++ b/NEWS.md @@ -1,3 +1,59 @@ +# uscogdata 0.4.0 + +## Cohorts can be named by predicate, not just by id + +`cog_spending()`, `cog_revenue()` and `cog_balances()` gain optional `state` +and `type` arguments. Both default to `NULL`, so every existing call behaves +exactly as before. + +Passing them expresses the cohort as a subquery against `canonical_fips_xwalk` +inside each statement, instead of round-tripping the ids through R and +rendering them back into a literal `IN` list: + +```r +# before: resolve 20,106 ids in R, then embed them in every statement +ids <- cog_gov_search(NULL, state = "CA", type = "city")$canonical_govid +cog_spending(ids, years = 2022) + +# now: the cohort never leaves the database +cog_spending(years = 2022, state = "CA", type = "city") +``` + +Measured against the production corpus, same FY2022 aggregate over the +20,106-government `type = "city"` cohort: + +| cohort expressed as | time | +|---|---:| +| `IN (20,106 literals)` | 449 ms | +| join against a temp cohort table | 99 ms | +| predicate on `canonical_fips_xwalk` | **94 ms** | +| no cohort filter at all (the floor) | 88 ms | + +**4.8x, within 7% of the floor.** The rendered `IN` list was 301,591 +characters and was re-parsed in 5-8 separate statements per call, so the cost +was paid repeatedly; the predicate's size is constant in the cohort. + +`state` and `type` use the same vocabulary and the same internal coercion as +`cog_gov_search()` -- `state` is a postal abbreviation (`"WI"`) even though the +crosswalk column holds a FIPS code (`"55"`). + +Supplying `govid` **and** `state`/`type` intersects them: the governments in +`govid` that also match the predicate. Naming no cohort at all now aborts with +class `uscogdata_no_cohort` rather than R's "argument is missing" error. + +When the cohort is named by predicate there is no id list to report, so +`provenance$scope$govids_found`/`govids_missing` are empty and +`provenance$scope$cohort` carries `state`, `type` and `n_governments` instead. +A `govid`-named cohort's provenance is unchanged. + +## Fixes + +* An unknown `state` abbreviation now aborts with "Unknown state abbreviation" + (class `uscogdata_unknown_state`) instead of base R's "subscript out of + bounds". `.state_abbrev_to_fips` is a named character vector, so `[[` on an + absent name threw before the curated message could be reached -- making that + message unreachable dead code in every verb that takes a `state`. + # uscogdata 0.3.0 First public release. diff --git a/R/balances.R b/R/balances.R index 994e0b2..26b439c 100644 --- a/R/balances.R +++ b/R/balances.R @@ -18,7 +18,9 @@ #' comparable to a GAAP fund balance from an ACFR. #' #' @param govid Canonical govid(s): a character vector, or a data frame with a -#' `canonical_govid` column (e.g. from [cog_gov_search()]). +#' `canonical_govid` column (e.g. from [cog_gov_search()]). `NULL` to name +#' the cohort by `state`/`type` instead. +#' @inheritParams cog_spending #' @param years Integer vector of fiscal years. #' @param category Optional character vector of categories to keep. One of #' `"Fund Balances"`, `"Insurance Trust Balances"`, @@ -63,15 +65,16 @@ #' requested years). `expenditure_concept`/`revenue_concept` are `NA` -- #' holdings are a stock, not a flow, so neither concept vocabulary applies. #' @export -cog_balances <- function(govid, years, category = NULL, +cog_balances <- function(govid = NULL, years, category = NULL, per_capita = FALSE, adjust_to_year = NULL, - basis = c("harmonized", "raw"), recipe = NULL) { + basis = c("harmonized", "raw"), recipe = NULL, + state = NULL, type = NULL) { call <- match.call() basis <- match.arg(basis, c("harmonized", "raw")) # Coerce FIRST, validate second: .validate_verb_inputs() asserts # is.character(govid), and a data-frame govid (cog_gov_search() output) has # not been unwrapped yet at this point. - govid <- .coerce_govid_input(govid) + govid <- if (is.null(govid)) NULL else .coerce_govid_input(govid) # The money verbs' validator, reused rather than re-implemented (R/spending.R). # It covers the exact superset cog_balances() needs -- including the # recipe/category mutual-exclusivity guard -- so a second local copy would @@ -91,6 +94,8 @@ cog_balances <- function(govid, years, category = NULL, years <- as.integer(years) if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year) + cohort <- .make_cohort(govid, state, type) + con <- .ensure_session() .require_balance_support(con) scope <- .check_govids_in_scope(govid) @@ -109,7 +114,7 @@ cog_balances <- function(govid, years, category = NULL, .validate_recipe_id(con, recipe) comps <- .recipe_components(con, recipe) recipe_label <- comps$label[[1]] - result <- .run_recipe(con, recipe, govid, years) + result <- .run_recipe(con, recipe, cohort, years) sql <- attr(result, "sql_query") result <- .shape_recipe_result(result, "balance_subtype", recipe_label) recipe_block <- list( @@ -119,7 +124,7 @@ cog_balances <- function(govid, years, category = NULL, category_for_prov <- recipe_label } else { sql <- .build_verb_sql("balance_annotated", "balance_subtype", - govid, years, category, + cohort, years, category, ig_view = NULL, subtype_scope = NULL) result <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) } @@ -127,7 +132,7 @@ cog_balances <- function(govid, years, category = NULL, # Order matters (matches .verb_spendrev()): per-capita first, so # .attach_real_dollars() deflates the nominal per-capita column into # amt_per_capita_real rather than needing amt_per_capita_nominal recomputed. - if (isTRUE(per_capita)) result <- .attach_per_capita(result, con, govid) + if (isTRUE(per_capita)) result <- .attach_per_capita(result, con) if (!is.null(adjust_to_year)) { result <- .attach_real_dollars(result, adjust_to_year, per_capita) } @@ -145,6 +150,7 @@ cog_balances <- function(govid, years, category = NULL, ) prov$scope$govids_found <- scope$found prov$scope$govids_missing <- scope$missing + prov$scope$cohort <- .cohort_provenance(con, cohort) prov$balance_caveats <- .balance_caveats( con, prov$codes_summed$observed, years diff --git a/R/basis.R b/R/basis.R index dd60a67..b173cbb 100644 --- a/R/basis.R +++ b/R/basis.R @@ -54,7 +54,7 @@ #' expenditure_concept = "total": ig_long_harmonized COALESCEs rather than #' drops NULL-harmonized rows, so harmonization never excludes an IG row. #' @noRd -.build_harmonization_block <- function(con, govid, years, resolved, +.build_harmonization_block <- function(con, cohort, years, resolved, subtype_col, subtype_scope) { if (!identical(resolved$basis, "harmonized")) { return(list( @@ -68,12 +68,12 @@ sql <- sprintf( "SELECT COUNT(*) AS n, COALESCE(SUM(amt), 0) * 1000.0 AS amt FROM long - WHERE canonical_govid IN (%s) AND year IN (%s) + WHERE %s AND year IN (%s) AND NOT is_aggregate AND harmonized_code IS NULL AND item_code IN ( SELECT item_code FROM summary_categories WHERE %s IN (%s) )", - .sql_lit_chr(govid), paste(as.integer(years), collapse = ","), + .cohort_sql(cohort), paste(as.integer(years), collapse = ","), subtype_col, .sql_lit_chr(subtype_scope) ) na <- DBI::dbGetQuery(con, sql) diff --git a/R/cohort.R b/R/cohort.R new file mode 100644 index 0000000..67fc16f --- /dev/null +++ b/R/cohort.R @@ -0,0 +1,127 @@ +# How a verb names the set of governments it queries. +# +# Historically there was one way: a `govid` character vector, rendered by +# .sql_lit_chr() into a quoted IN list. That is fine for a handful of +# governments and pathological for a fleet. Measured against the production +# corpus, the same FY2022 aggregate over the 20,106-government `type = "city"` +# cohort: +# +# cohort expressed as time +# IN (20,106 literals) 449 ms +# join against a temp cohort table 99 ms +# predicate on canonical_fips_xwalk 94 ms +# no cohort filter at all (the floor) 88 ms +# +# 4.8x, and within 7% of the no-filter floor. The rendered IN list is 301,591 +# characters and .verb_spendrev() embeds it in 5-8 separate statements per +# call, so the parse-and-plan cost is paid over and over (uscogdata#58). +# +# A cohort therefore has two independent halves, and a query can carry either +# or both: +# +# ids an explicit canonical_govid vector -> literal IN list +# predicate state/type over canonical_fips_xwalk -> IN (SELECT ...) +# +# Both together is an INTERSECTION -- "these ids, narrowed to that state/type" +# -- never a precedence rule where one silently wins. + +#' Build the internal cohort object shared by every query verb. +#' +#' `state` and `type` are coerced with the SAME helpers `cog_gov_search()` +#' uses. That is load-bearing, not tidiness: the public argument is a postal +#' abbreviation (`"WI"`) while `canonical_fips_xwalk.fips_state` holds a FIPS +#' code (`"55"`), and `type` is a label (`"city"`) against an integer +#' `govs_type`. A predicate written against the raw parameter matches nothing +#' and returns an empty result indistinguishable from "this government +#' reported nothing" -- cog-api hit exactly that trap optimizing this path. +#' One definition of the translation, not two. +#' +#' @param govid Already-coerced character vector of canonical_govids, or NULL. +#' @param state Postal abbreviation or FIPS code, or NULL. +#' @param type Type label or integer code, or NULL. +#' @noRd +.make_cohort <- function(govid = NULL, state = NULL, type = NULL) { + if (is.null(govid) && is.null(state) && is.null(type)) { + cli::cli_abort(c( + "A cohort must be named.", + "*" = "Pass {.arg govid} for specific governments, or {.arg state}/{.arg type} for every government matching a predicate.", + "i" = "Passing both intersects them: the governments in {.arg govid} that also match {.arg state}/{.arg type}." + ), class = "uscogdata_no_cohort") + } + structure( + list( + ids = govid, + state = state, + type = type, + state_fips = if (is.null(state)) NULL else .coerce_state_to_fips(state), + type_int = if (is.null(type)) NULL else .coerce_type(type) + ), + class = "uscogdata_cohort" + ) +} + +#' Is any part of this cohort expressed as an xwalk predicate? +#' @noRd +.cohort_by_predicate <- function(cohort) { + !is.null(cohort$state_fips) || !is.null(cohort$type_int) +} + +#' Render the cohort as a SQL boolean expression over `col`. +#' +#' `col` may be qualified (`"l.canonical_govid"`, `"x.canonical_govid"`) -- +#' several call sites join the xwalk under an alias. The subquery's own +#' projected column stays unqualified: it selects from canonical_fips_xwalk, +#' not from the outer relation. +#' @noRd +.cohort_sql <- function(cohort, col = "canonical_govid") { + preds <- character(0) + + if (!is.null(cohort$ids)) { + preds <- c(preds, sprintf("%s IN (%s)", col, .sql_lit_chr(cohort$ids))) + } + + if (.cohort_by_predicate(cohort)) { + xwalk_preds <- character(0) + if (!is.null(cohort$state_fips)) { + xwalk_preds <- c(xwalk_preds, + sprintf("fips_state = %s", .sql_lit_chr(cohort$state_fips))) + } + if (!is.null(cohort$type_int)) { + xwalk_preds <- c(xwalk_preds, sprintf("govs_type = %d", cohort$type_int)) + } + preds <- c(preds, sprintf( + "%s IN (SELECT canonical_govid FROM canonical_fips_xwalk WHERE %s)", + col, paste(xwalk_preds, collapse = " AND ") + )) + } + + paste(preds, collapse = " AND ") +} + +#' How many governments the cohort covers. +#' +#' One COUNT against the crosswalk, used only to populate the provenance +#' `scope$cohort` block. Deliberately a count rather than the id list: a +#' fleet-scale cohort would otherwise put 20,000 ids into every response body, +#' which is the cost this issue exists to remove. +#' @noRd +.cohort_count <- function(con, cohort) { + sql <- sprintf( + "SELECT COUNT(*) AS n FROM canonical_fips_xwalk WHERE %s", + .cohort_sql(cohort) + ) + as.integer(DBI::dbGetQuery(con, sql)$n[[1]]) +} + +#' The provenance `scope$cohort` block for a predicate cohort, or NULL when +#' the cohort was named by id alone (in which case `govids_found`/ +#' `govids_missing` already describe it exactly). +#' @noRd +.cohort_provenance <- function(con, cohort) { + if (!.cohort_by_predicate(cohort)) return(NULL) + list( + state = if (is.null(cohort$state)) NA_character_ else as.character(cohort$state), + type = if (is.null(cohort$type)) NA_character_ else as.character(cohort$type), + n_governments = .cohort_count(con, cohort) + ) +} diff --git a/R/complete.R b/R/complete.R index a5cfc77..5b16b68 100644 --- a/R/complete.R +++ b/R/complete.R @@ -57,7 +57,7 @@ #' aggregate rows. Without it the grid would offer cells the verb structurally #' never returns, so every one of them would fill as a phantom $0. #' @noRd -.completion_grid_sql <- function(subtype_col, govid, years, category, +.completion_grid_sql <- function(subtype_col, cohort, years, category, subtype_scope) { category_pred <- if (is.null(category)) { "" @@ -76,13 +76,13 @@ JOIN canonical_fips_xwalk x ON x.govs_type = cs.type JOIN summary_categories c ON c.item_code = cs.item_code JOIN representation r ON r.year = cs.year - WHERE x.canonical_govid IN (%2$s) + WHERE %2$s AND cs.year IN (%3$s) AND NOT cs.is_aggregate AND c.category IS NOT NULL AND c.%1$s IN (%4$s) %5$s", - subtype_col, .sql_lit_chr(govid), + subtype_col, .cohort_sql(cohort, "x.canonical_govid"), paste(as.integer(years), collapse = ","), .sql_lit_chr(subtype_scope), category_pred ) @@ -94,10 +94,10 @@ #' provenance block. Reported rows are passed through untouched -- filling #' must never alter or drop what the corpus actually published. #' @noRd -.complete_result <- function(result, con, subtype_col, govid, years, category, +.complete_result <- function(result, con, subtype_col, cohort, years, category, subtype_scope) { grid <- tibble::as_tibble(DBI::dbGetQuery( - con, .completion_grid_sql(subtype_col, govid, years, category, subtype_scope) + con, .completion_grid_sql(subtype_col, cohort, years, category, subtype_scope) )) result$value_source <- rep("reported", nrow(result)) diff --git a/R/recipes.R b/R/recipes.R index 11a55be..15a1c6d 100644 --- a/R/recipes.R +++ b/R/recipes.R @@ -113,7 +113,7 @@ cog_recipes <- function(pattern = NULL) { #' `year_min`/`year_max` -- so there is no double-counting. (Checkpoint #' review docs/phase_r_harmonization_review.md ยง 0.2.) #' @noRd -.run_recipe <- function(con, recipe_id, govid, years) { +.run_recipe <- function(con, recipe_id, cohort, years) { sql <- sprintf( "SELECT l.year, l.canonical_govid, COALESCE(x.gov_name, l.gov_name) AS gov_name, @@ -128,11 +128,11 @@ cog_recipes <- function(pattern = NULL) { OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3)) LEFT JOIN canonical_fips_xwalk x USING (canonical_govid) WHERE r.recipe_id = %1$s - AND l.canonical_govid IN (%2$s) + AND %2$s AND l.year IN (%3$s) GROUP BY 1, 2, 3 ORDER BY 1, 2", - .sql_lit_chr(recipe_id), .sql_lit_chr(govid), + .sql_lit_chr(recipe_id), .cohort_sql(cohort, "l.canonical_govid"), paste(as.integer(years), collapse = ",") ) result <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) diff --git a/R/revenue.R b/R/revenue.R index 376f9a6..b6efe03 100644 --- a/R/revenue.R +++ b/R/revenue.R @@ -50,11 +50,12 @@ #' optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`, #' and `value_source` when `complete = TRUE`. #' @export -cog_revenue <- function(govid, years, category = NULL, +cog_revenue <- function(govid = NULL, years, category = NULL, per_capita = FALSE, adjust_to_year = NULL, basis = c("harmonized", "raw"), recipe = NULL, revenue_concept = c("general", "total"), - complete = FALSE, limit = NULL, offset = NULL) { + complete = FALSE, limit = NULL, offset = NULL, + state = NULL, type = NULL) { # flow_prefixes no longer classifies rows (crosswalk revenue_subtype # membership does -- General Revenue, i.e. everything except # insurance_trust) -- it only scopes the recipe-suggestion machinery to @@ -75,6 +76,8 @@ cog_revenue <- function(govid, years, category = NULL, revenue_concept = revenue_concept, complete = complete, limit = limit, - offset = offset + offset = offset, + state = state, + type = type ) } diff --git a/R/search.R b/R/search.R index 01c27bd..cb3abbd 100644 --- a/R/search.R +++ b/R/search.R @@ -184,11 +184,18 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { if (!is.character(state) || length(state) != 1L) { cli::cli_abort("`state` must be a 2-letter USPS abbrev or a FIPS integer.") } - fips <- .state_abbrev_to_fips[[toupper(state)]] - if (is.null(fips)) { - cli::cli_abort("Unknown state abbreviation: {state}.") + # Membership tested before the lookup, not after: `.state_abbrev_to_fips` is + # a named CHARACTER vector, and `[[` on a name it does not carry throws + # base R's "subscript out of bounds" rather than returning NULL -- which + # made the curated message below unreachable dead code. Reported as a bare + # subscript error, `cog_gov_search(state = "ZZ")` gave no hint that the + # argument wants a postal abbreviation. + key <- toupper(state) + if (!key %in% names(.state_abbrev_to_fips)) { + cli::cli_abort("Unknown state abbreviation: {state}.", + class = "uscogdata_unknown_state") } - fips + .state_abbrev_to_fips[[key]] } # USPS state / territory abbreviation -> 2-digit FIPS code. diff --git a/R/spending.R b/R/spending.R index 61a6559..40407a2 100644 --- a/R/spending.R +++ b/R/spending.R @@ -68,7 +68,9 @@ #' millions/billions). The conversion is recorded in the provenance attribute #' under `transformations$units_conversion`. #' -#' @param govid Character vector of `canonical_govid` values. +#' @param govid Character vector of `canonical_govid` values, or `NULL` to name +#' the cohort by `state`/`type` instead. One of `govid`, `state`, or `type` +#' is required. #' @param years Integer vector of years. #' @param category Character vector of category names (from #' `summary_categories.category`), or `NULL` for all categories broken out @@ -185,6 +187,29 @@ #' `recipe` and with `complete = TRUE` -- see `offset` and `total_rows`. #' @param offset Rows to skip before `limit` starts counting (0-based). #' Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set. +#' @param state,type Name the cohort by predicate instead of by id: `state` is +#' a 2-letter USPS abbreviation (or a FIPS code) and `type` is one of +#' `"state"`, `"county"`, `"city"`, `"township"` (or the integer `0:3`) -- +#' the same vocabulary, and the same internal coercion, as +#' [cog_gov_search()]. Both default to `NULL`. +#' +#' The cohort is then expressed as a subquery against `canonical_fips_xwalk` +#' inside each statement rather than round-tripped through R as a literal id +#' list. For a fleet-scale cohort that is the difference between a +#' 301,591-character `IN` list re-parsed in 5--8 statements per call and a +#' constant-size predicate: measured at **94 ms versus 449 ms** for the same +#' FY2022 aggregate over the 20,106-government `type = "city"` cohort, within +#' 7% of the no-filter floor. +#' +#' Supplying `govid` **and** `state`/`type` INTERSECTS them -- the +#' governments in `govid` that also match the predicate -- rather than one +#' silently taking precedence. Naming no cohort at all (`govid`, `state` and +#' `type` all `NULL`) aborts with class `uscogdata_no_cohort`. +#' +#' When the cohort is named by predicate, `provenance$scope$govids_found` +#' and `govids_missing` are empty -- there is no id list to report against -- +#' and `provenance$scope$cohort` carries `state`, `type` and +#' `n_governments` instead. A `govid`-named cohort reports exactly as before. #' @return Tibble with columns `year`, `canonical_govid`, `gov_name`, #' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real`, #' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`, @@ -198,11 +223,12 @@ #' round trip -- so a caller walking pages never has to ask "how many are #' there" separately. #' @export -cog_spending <- function(govid, years, category = NULL, +cog_spending <- function(govid = NULL, years, category = NULL, per_capita = FALSE, adjust_to_year = NULL, basis = c("harmonized", "raw"), recipe = NULL, expenditure_concept = c("primary", "direct", "total"), - complete = FALSE, limit = NULL, offset = NULL) { + complete = FALSE, limit = NULL, offset = NULL, + state = NULL, type = NULL) { # flow_prefixes no longer classifies rows (crosswalk subtype membership # does, per expenditure_concept) -- it only scopes the recipe-suggestion # machinery to this verb's recipe families (see R/suggestions.R; the @@ -223,7 +249,9 @@ cog_spending <- function(govid, years, category = NULL, expenditure_concept = expenditure_concept, complete = complete, limit = limit, - offset = offset + offset = offset, + state = state, + type = type ) } @@ -251,7 +279,8 @@ cog_spending <- function(govid, years, category = NULL, basis = c("harmonized", "raw"), recipe = NULL, expenditure_concept = c("primary", "direct", "total"), revenue_concept = c("general", "total"), - complete = FALSE, limit = NULL, offset = NULL) { + complete = FALSE, limit = NULL, offset = NULL, + state = NULL, type = NULL) { basis_explicit <- length(basis) == 1L basis <- match.arg(basis, c("harmonized", "raw")) # match.arg() itself throws a base `simpleError`, not an rlang-classed @@ -291,7 +320,10 @@ cog_spending <- function(govid, years, category = NULL, .revenue_concept_subtypes(revenue_concept) } - govid <- .coerce_govid_input(govid, arg = "govid") + # NULL `govid` means "the cohort is named by predicate"; anything else is + # coerced and validated exactly as before, so an empty or wrong-typed vector + # still fails with its original message rather than being read as absent. + govid <- if (is.null(govid)) NULL else .coerce_govid_input(govid, arg = "govid") # allow_all_categories = TRUE: cog_spending()/cog_revenue() are the two # verbs the reserved pseudo-category is defined for. cog_balances() shares # this validator but leaves the argument at its FALSE default, so it @@ -300,6 +332,10 @@ cog_spending <- function(govid, years, category = NULL, .validate_verb_inputs(govid, years, category, per_capita, adjust_to_year, recipe, allow_all_categories = TRUE) + # Built after validation so the argument-shape errors above keep firing + # first, and before .ensure_session() so a bad state/type costs no I/O. + cohort <- .make_cohort(govid, state, type) + # Recognize the reserved pseudo-category. Detected after type validation so a # non-character `category` still fails with the ordinary type error. all_categories <- !is.null(category) && .ALL_CATEGORIES %in% category @@ -407,7 +443,7 @@ cog_spending <- function(govid, years, category = NULL, .validate_recipe_id(con, recipe) comps <- .recipe_components(con, recipe) recipe_label <- comps$label[[1]] - result <- .run_recipe(con, recipe, govid, years) + result <- .run_recipe(con, recipe, cohort, years) sql <- attr(result, "sql_query") result <- .shape_recipe_result(result, subtype_col, recipe_label) recipe_block <- list( @@ -423,7 +459,7 @@ cog_spending <- function(govid, years, category = NULL, } else { NULL } - sql <- .build_verb_sql(view, subtype_col, govid, years, + sql <- .build_verb_sql(view, subtype_col, cohort, years, if (all_categories) NULL else category, ig_view, subtype_scope, all_categories = all_categories, @@ -441,7 +477,7 @@ cog_spending <- function(govid, years, category = NULL, } else { count_sql <- sprintf( "SELECT COUNT(*) AS n FROM (%s) AS _uncounted", - .build_verb_sql(view, subtype_col, govid, years, + .build_verb_sql(view, subtype_col, cohort, years, if (all_categories) NULL else category, ig_view, subtype_scope, all_categories = all_categories) @@ -457,13 +493,13 @@ cog_spending <- function(govid, years, category = NULL, # spurious 0. completion <- list(applied = FALSE, rows_filled = 0L, absence_means = list()) if (complete) { - result <- .complete_result(result, con, subtype_col, govid, years, + result <- .complete_result(result, con, subtype_col, cohort, years, category, subtype_scope) completion <- attr(result, ".completion") attr(result, ".completion") <- NULL } - if (per_capita) result <- .attach_per_capita(result, con, govid) + if (per_capita) result <- .attach_per_capita(result, con) if (!is.null(adjust_to_year)) { result <- .attach_real_dollars(result, adjust_to_year, per_capita) } @@ -491,7 +527,7 @@ cog_spending <- function(govid, years, category = NULL, basis_for_prov <- resolved$basis basis_note_for_prov <- resolved$note harmonization <- .build_harmonization_block( - con, govid, years, resolved, subtype_col, subtype_scope + con, cohort, years, resolved, subtype_col, subtype_scope ) # C1(a): gap detection must run against the Direct leg alone. `result` # can also carry UNION'd intergovernmental rows (expenditure_concept = @@ -506,7 +542,7 @@ cog_spending <- function(govid, years, category = NULL, } else { result } - suggestions <- .build_suggestions(con, govid, years, category, + suggestions <- .build_suggestions(con, cohort, years, category, direct_leg_result, resolved$basis, flow_prefixes, .select_long_view(view_base, resolved$basis), @@ -611,6 +647,12 @@ cog_spending <- function(govid, years, category = NULL, ) prov$scope$govids_found <- scope$found prov$scope$govids_missing <- scope$missing + # A predicate-named cohort has no id list to report found/missing against + # (both stay empty), so it describes itself instead. Deliberately a COUNT + # rather than the resolved ids: enumerating them would put 20,000 govids in + # every fleet-scale response body, which is the cost this path exists to + # remove. NULL for a govid-named cohort, so that output is untouched. + prov$scope$cohort <- .cohort_provenance(con, cohort) attr(result, "provenance") <- prov attr(result, ".popyear_range") <- NULL # Attached here, after every downstream transform (per_capita/real-dollar @@ -639,7 +681,11 @@ cog_spending <- function(govid, years, category = NULL, .validate_verb_inputs <- function(govid, years, category, per_capita, adjust_to_year, recipe = NULL, allow_all_categories = FALSE) { - if (!is.character(govid) || length(govid) == 0L) { + # NULL is allowed only because the caller has already established that the + # cohort is named some other way (`state`/`type`); .make_cohort() is what + # refuses a call that names no cohort at all. A supplied-but-empty `govid` + # still fails here, exactly as before. + if (!is.null(govid) && (!is.character(govid) || length(govid) == 0L)) { cli::cli_abort("`govid` must be a non-empty character vector.") } if (!(is.integer(years) || is.numeric(years)) || length(years) == 0L) { @@ -740,10 +786,10 @@ cog_spending <- function(govid, years, category = NULL, } #' @noRd -.build_verb_sql <- function(view, subtype_col, govid, years, category, +.build_verb_sql <- function(view, subtype_col, cohort, years, category, ig_view = NULL, subtype_scope = NULL, all_categories = FALSE, limit = NULL, offset = NULL) { - govid_lit <- .sql_lit_chr(govid) + cohort_pred <- .cohort_sql(cohort) years_lit <- paste(as.integer(years), collapse = ",") # In all-categories mode there is no category filter: the sum is defined by # the concept's SUBTYPE allowlist (subtype_pred below), which is the real @@ -813,13 +859,13 @@ cog_spending <- function(govid, years, category = NULL, string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS codes_included, bool_or(is_aggregate) AS aggregate_fallback FROM %2$s - WHERE canonical_govid IN (%3$s) + WHERE %3$s AND year IN (%4$s) %5$s %6$s GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, %1$s%8$s ORDER BY year, canonical_govid, %1$s%8$s", - subtype_col, source_expr, govid_lit, years_lit, category_pred, subtype_pred, + subtype_col, source_expr, cohort_pred, years_lit, category_pred, subtype_pred, category_select, category_group ) @@ -845,8 +891,16 @@ cog_spending <- function(govid, years, category = NULL, } } +#' Join population onto a result and derive the per-capita columns. +#' +#' The population lookup is keyed on the govids PRESENT IN `result`, not on the +#' cohort that produced it. Those are the only ones the LEFT JOIN below can +#' match, so the joined output is identical either way -- but it means this +#' works unchanged for a cohort named by predicate (where no id list exists in +#' R at all), and on a paginated call it looks up one page's governments +#' instead of the whole fleet's. #' @noRd -.attach_per_capita <- function(result, con, govid) { +.attach_per_capita <- function(result, con) { if (nrow(result) == 0L) { result$amt_per_capita_nominal <- numeric(0) result$pop_source <- character(0) @@ -859,7 +913,7 @@ cog_spending <- function(govid, years, category = NULL, FROM gov_population_yearly WHERE canonical_govid IN (%s) AND year IN (%s)", - .sql_lit_chr(govid), years_lit + .sql_lit_chr(unique(result$canonical_govid)), years_lit ) pops <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) result <- dplyr::left_join(result, pops, diff --git a/R/suggestions.R b/R/suggestions.R index b2b0ce4..f089040 100644 --- a/R/suggestions.R +++ b/R/suggestions.R @@ -42,8 +42,8 @@ #' verb call: recipes whose generic join would fill a real gap in `result`. #' #' @param con Active DuckDB connection. -#' @param govid Character vector of canonical_govid values (the verb's raw -#' `govid`). +#' @param cohort The verb's cohort object (see `.make_cohort()`), naming the +#' governments by id, by state/type predicate, or both. #' @param years Integer vector of requested years. #' @param category `category` argument as passed to the verb (character #' vector or `NULL`; suggestions are only computed when non-NULL). @@ -85,7 +85,7 @@ #' ig_recipe_id, trigger, suppressed_amount, suppressed_years, #' suppressed_codes)`, possibly empty. #' @noRd -.build_suggestions <- function(con, govid, years, category, result, basis, +.build_suggestions <- function(con, cohort, years, category, result, basis, flow_prefixes, long_view, all_categories = FALSE, subtype_col = NULL, subtype_scope = NULL) { @@ -160,7 +160,7 @@ # violation would kill signposting, the exact failure class uscogdata#9 # exists to prevent. Owner's call: keep this simple; a batch-aware # optimization, if one is worth building, is a separate issue. - supp <- .suppressed_components(con, candidates, govid, years, long_view, flow_prefixes) + supp <- .suppressed_components(con, candidates, cohort, years, long_view, flow_prefixes) if (length(gap_years) == 0L && nrow(supp) == 0L) return(list()) @@ -188,9 +188,9 @@ OR (r.gov_type_scope = 'state' AND l.type = 0) OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3)) WHERE r.recipe_id IN (%s) - AND l.canonical_govid IN (%s) + AND %s AND l.year IN (%s)", - .sql_lit_chr(candidates), .sql_lit_chr(govid), + .sql_lit_chr(candidates), .cohort_sql(cohort, "l.canonical_govid"), paste(gap_years, collapse = ",") )) } diff --git a/R/suppression.R b/R/suppression.R index e9a01b4..c822fc0 100644 --- a/R/suppression.R +++ b/R/suppression.R @@ -50,7 +50,9 @@ #' #' @param con Active DuckDB connection. #' @param candidates Character vector of recipe ids to measure. -#' @param govid Character vector of canonical_govid values. +#' @param cohort The verb's cohort object (see `.make_cohort()`), rendered +#' into the govid predicate on both the outer scan and the restated +#' NOT EXISTS filter. #' @param years Integer vector of requested years. #' @param long_view Name of the verb's long view, from `.select_long_view()`. #' @param flow_prefixes The calling verb's own flow-type prefixes (see @@ -60,7 +62,7 @@ #' dollars), `suppressed_codes` (comma-joined, sorted). Zero rows when #' nothing is suppressed. #' @noRd -.suppressed_components <- function(con, candidates, govid, years, long_view, +.suppressed_components <- function(con, candidates, cohort, years, long_view, flow_prefixes) { empty <- tibble::tibble( recipe_id = character(0), year = numeric(0), @@ -93,7 +95,7 @@ OR (r.gov_type_scope = 'state' AND l.type = 0) OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3)) WHERE r.recipe_id IN (%1$s) - AND l.canonical_govid IN (%2$s) + AND %2$s AND l.year IN (%3$s) AND l.amt <> 0 AND LEFT(r.component_code, 1) IN (%5$s) @@ -103,13 +105,13 @@ AND v.year = l.year AND v.item_code = l.item_code AND v.year IN (%3$s) -- restated: enables partition pruning (I3a) - AND v.canonical_govid IN (%2$s) -- restated: pushes the govid filter (I3a) + AND %6$s -- restated: pushes the cohort filter (I3a) ) GROUP BY 1, 2 ORDER BY 1, 2", - .sql_lit_chr(candidates), .sql_lit_chr(govid), + .sql_lit_chr(candidates), .cohort_sql(cohort, "l.canonical_govid"), paste(as.integer(years), collapse = ","), long_view, - .sql_lit_chr(flow_prefixes) + .sql_lit_chr(flow_prefixes), .cohort_sql(cohort, "v.canonical_govid") ) tibble::as_tibble(DBI::dbGetQuery(con, sql)) } diff --git a/man/cog_balances.Rd b/man/cog_balances.Rd index 4dab4a7..e573552 100644 --- a/man/cog_balances.Rd +++ b/man/cog_balances.Rd @@ -5,18 +5,21 @@ \title{Cash and security holdings for one or more governments} \usage{ cog_balances( - govid, + govid = NULL, years, category = NULL, per_capita = FALSE, adjust_to_year = NULL, basis = c("harmonized", "raw"), - recipe = NULL + recipe = NULL, + state = NULL, + type = NULL ) } \arguments{ \item{govid}{Canonical govid(s): a character vector, or a data frame with a -`canonical_govid` column (e.g. from [cog_gov_search()]).} +`canonical_govid` column (e.g. from [cog_gov_search()]). `NULL` to name +the cohort by `state`/`type` instead.} \item{years}{Integer vector of fiscal years.} @@ -49,6 +52,30 @@ and raw space are identical for holdings. Reported in \item{recipe}{Optional harmonization recipe id (see [cog_recipes()]). `"cash_securities_z77_wide"` and `"cash_securities_z78_wide"` bridge the wide era to the modern one.} + +\item{state, type}{Name the cohort by predicate instead of by id: `state` is + a 2-letter USPS abbreviation (or a FIPS code) and `type` is one of + `"state"`, `"county"`, `"city"`, `"township"` (or the integer `0:3`) -- + the same vocabulary, and the same internal coercion, as + [cog_gov_search()]. Both default to `NULL`. + + The cohort is then expressed as a subquery against `canonical_fips_xwalk` + inside each statement rather than round-tripped through R as a literal id + list. For a fleet-scale cohort that is the difference between a + 301,591-character `IN` list re-parsed in 5--8 statements per call and a + constant-size predicate: measured at **94 ms versus 449 ms** for the same + FY2022 aggregate over the 20,106-government `type = "city"` cohort, within + 7% of the no-filter floor. + + Supplying `govid` **and** `state`/`type` INTERSECTS them -- the + governments in `govid` that also match the predicate -- rather than one + silently taking precedence. Naming no cohort at all (`govid`, `state` and + `type` all `NULL`) aborts with class `uscogdata_no_cohort`. + + When the cohort is named by predicate, `provenance$scope$govids_found` + and `govids_missing` are empty -- there is no id list to report against -- + and `provenance$scope$cohort` carries `state`, `type` and + `n_governments` instead. A `govid`-named cohort reports exactly as before.} } \value{ Tibble with columns `year`, `canonical_govid`, `gov_name`, diff --git a/man/cog_revenue.Rd b/man/cog_revenue.Rd index 940fe20..e94a3d8 100644 --- a/man/cog_revenue.Rd +++ b/man/cog_revenue.Rd @@ -5,7 +5,7 @@ \title{Summarized revenue by category} \usage{ cog_revenue( - govid, + govid = NULL, years, category = NULL, per_capita = FALSE, @@ -15,11 +15,15 @@ cog_revenue( revenue_concept = c("general", "total"), complete = FALSE, limit = NULL, - offset = NULL + offset = NULL, + state = NULL, + type = NULL ) } \arguments{ -\item{govid}{Character vector of `canonical_govid` values.} +\item{govid}{Character vector of `canonical_govid` values, or `NULL` to name +the cohort by `state`/`type` instead. One of `govid`, `state`, or `type` +is required.} \item{years}{Integer vector of years.} @@ -126,6 +130,30 @@ exactly as before this parameter existed. Mutually exclusive with \item{offset}{Rows to skip before `limit` starts counting (0-based). Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set.} + +\item{state, type}{Name the cohort by predicate instead of by id: `state` is + a 2-letter USPS abbreviation (or a FIPS code) and `type` is one of + `"state"`, `"county"`, `"city"`, `"township"` (or the integer `0:3`) -- + the same vocabulary, and the same internal coercion, as + [cog_gov_search()]. Both default to `NULL`. + + The cohort is then expressed as a subquery against `canonical_fips_xwalk` + inside each statement rather than round-tripped through R as a literal id + list. For a fleet-scale cohort that is the difference between a + 301,591-character `IN` list re-parsed in 5--8 statements per call and a + constant-size predicate: measured at **94 ms versus 449 ms** for the same + FY2022 aggregate over the 20,106-government `type = "city"` cohort, within + 7% of the no-filter floor. + + Supplying `govid` **and** `state`/`type` INTERSECTS them -- the + governments in `govid` that also match the predicate -- rather than one + silently taking precedence. Naming no cohort at all (`govid`, `state` and + `type` all `NULL`) aborts with class `uscogdata_no_cohort`. + + When the cohort is named by predicate, `provenance$scope$govids_found` + and `govids_missing` are empty -- there is no id list to report against -- + and `provenance$scope$cohort` carries `state`, `type` and + `n_governments` instead. A `govid`-named cohort reports exactly as before.} } \value{ Tibble with columns `year`, `canonical_govid`, `gov_name`, diff --git a/man/cog_spending.Rd b/man/cog_spending.Rd index f1374e8..9738687 100644 --- a/man/cog_spending.Rd +++ b/man/cog_spending.Rd @@ -5,7 +5,7 @@ \title{Summarized spending by category} \usage{ cog_spending( - govid, + govid = NULL, years, category = NULL, per_capita = FALSE, @@ -15,11 +15,15 @@ cog_spending( expenditure_concept = c("primary", "direct", "total"), complete = FALSE, limit = NULL, - offset = NULL + offset = NULL, + state = NULL, + type = NULL ) } \arguments{ -\item{govid}{Character vector of `canonical_govid` values.} +\item{govid}{Character vector of `canonical_govid` values, or `NULL` to name +the cohort by `state`/`type` instead. One of `govid`, `state`, or `type` +is required.} \item{years}{Integer vector of years.} @@ -146,6 +150,30 @@ exactly as before this parameter existed. Mutually exclusive with \item{offset}{Rows to skip before `limit` starts counting (0-based). Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set.} + +\item{state, type}{Name the cohort by predicate instead of by id: `state` is + a 2-letter USPS abbreviation (or a FIPS code) and `type` is one of + `"state"`, `"county"`, `"city"`, `"township"` (or the integer `0:3`) -- + the same vocabulary, and the same internal coercion, as + [cog_gov_search()]. Both default to `NULL`. + + The cohort is then expressed as a subquery against `canonical_fips_xwalk` + inside each statement rather than round-tripped through R as a literal id + list. For a fleet-scale cohort that is the difference between a + 301,591-character `IN` list re-parsed in 5--8 statements per call and a + constant-size predicate: measured at **94 ms versus 449 ms** for the same + FY2022 aggregate over the 20,106-government `type = "city"` cohort, within + 7% of the no-filter floor. + + Supplying `govid` **and** `state`/`type` INTERSECTS them -- the + governments in `govid` that also match the predicate -- rather than one + silently taking precedence. Naming no cohort at all (`govid`, `state` and + `type` all `NULL`) aborts with class `uscogdata_no_cohort`. + + When the cohort is named by predicate, `provenance$scope$govids_found` + and `govids_missing` are empty -- there is no id list to report against -- + and `provenance$scope$cohort` carries `state`, `type` and + `n_governments` instead. A `govid`-named cohort reports exactly as before.} } \value{ Tibble with columns `year`, `canonical_govid`, `gov_name`, diff --git a/tests/testthat/test-all-categories.R b/tests/testthat/test-all-categories.R index a8ceb08..0052692 100644 --- a/tests/testthat/test-all-categories.R +++ b/tests/testthat/test-all-categories.R @@ -4,7 +4,7 @@ test_that(".build_verb_sql emits a literal category and no category filter in al sql <- uscogdata:::.build_verb_sql( view = "spending_annotated", subtype_col = "spend_subtype", - govid = "552025209777", + cohort = uscogdata:::.make_cohort("552025209777"), years = 2019L, category = NULL, subtype_scope = c("operations", "capital"), @@ -24,7 +24,7 @@ test_that(".build_verb_sql emits a literal category and no category filter in al test_that(".build_verb_sql is unchanged when all_categories is FALSE", { args <- list( view = "spending_annotated", subtype_col = "spend_subtype", - govid = "552025209777", years = 2019L, category = NULL, + cohort = uscogdata:::.make_cohort("552025209777"), years = 2019L, category = NULL, subtype_scope = c("operations", "capital") ) old <- do.call(uscogdata:::.build_verb_sql, args) @@ -244,7 +244,7 @@ test_that('"All Categories" candidate scoping is symmetric with .build_verb_sql( con <- uscogdata:::.ensure_session() none <- uscogdata:::.build_suggestions( - con, govid = "010000226085", years = 2011L, + con, cohort = uscogdata:::.make_cohort("010000226085"), years = 2011L, category = "All Categories", result = NULL, basis = "harmonized", flow_prefixes = c("E", "F", "G"), long_view = "spending_long_harmonized", @@ -255,7 +255,7 @@ test_that('"All Categories" candidate scoping is symmetric with .build_verb_sql( expect_length(none, 0L) scoped <- uscogdata:::.build_suggestions( - con, govid = "010000226085", years = 2011L, + con, cohort = uscogdata:::.make_cohort("010000226085"), years = 2011L, category = "All Categories", result = NULL, basis = "harmonized", flow_prefixes = c("E", "F", "G"), long_view = "spending_long_harmonized", diff --git a/tests/testthat/test-cohort-verbs.R b/tests/testthat/test-cohort-verbs.R new file mode 100644 index 0000000..3b6f655 --- /dev/null +++ b/tests/testthat/test-cohort-verbs.R @@ -0,0 +1,218 @@ +# The money verbs accepting a cohort by predicate (uscogdata#58). +# +# The load-bearing property is EQUIVALENCE: naming the same set of governments +# by id and by state/type must return the same rows. Everything else here is +# about the ways that equivalence could silently break -- the postal/FIPS +# translation, the intersection rule, and provenance no longer having an id +# list to describe. +# +# Fixture cohorts used: RI (fips 44) cities = 8 governments, DE (fips 10) +# counties = 3. Small on purpose; the size of the win is measured against the +# production corpus, not here. + +# --- equivalence ----------------------------------------------------------- + +test_that("a predicate cohort returns exactly what the same ids return", { + skip_if_no_corpus() + with_fixture_corpus({ + ids <- cog_gov_search(NULL, state = "RI", type = "city")$canonical_govid + expect_gt(length(ids), 1L) + + by_id <- cog_spending(govid = ids, years = 2019) + by_pred <- cog_spending(years = 2019, state = "RI", type = "city") + + # Compare the data itself, ignoring the provenance attribute -- which is + # SUPPOSED to differ (see the scope tests below). + expect_equal( + as.data.frame(by_id[order(by_id$canonical_govid, by_id$category), ]), + as.data.frame(by_pred[order(by_pred$canonical_govid, by_pred$category), ]), + ignore_attr = TRUE + ) + expect_gt(nrow(by_pred), 0L) + }) +}) + +test_that("equivalence holds for cog_revenue()", { + skip_if_no_corpus() + with_fixture_corpus({ + ids <- cog_gov_search(NULL, state = "DE", type = "county")$canonical_govid + by_id <- cog_revenue(govid = ids, years = 2019) + by_pred <- cog_revenue(years = 2019, state = "DE", type = "county") + expect_equal(nrow(by_id), nrow(by_pred)) + expect_equal(sum(by_id$amt_nominal), sum(by_pred$amt_nominal)) + }) +}) + +test_that("equivalence holds for cog_balances()", { + skip_if_no_corpus() + with_fixture_corpus({ + ids <- cog_gov_search(NULL, state = "DE", type = "county")$canonical_govid + by_id <- cog_balances(govid = ids, years = 2019) + by_pred <- cog_balances(years = 2019, state = "DE", type = "county") + expect_equal(nrow(by_id), nrow(by_pred)) + expect_equal(sum(by_id$amt_nominal), sum(by_pred$amt_nominal)) + }) +}) + +test_that("equivalence survives per_capita, adjust_to_year and pagination", { + skip_if_no_corpus() + with_fixture_corpus({ + ids <- cog_gov_search(NULL, state = "RI", type = "city")$canonical_govid + + # per_capita now keys its population lookup on the rows in the result + # rather than on the requested cohort; these must stay identical. + by_id <- cog_spending(govid = ids, years = 2019, per_capita = TRUE, + adjust_to_year = 2020) + by_pred <- cog_spending(years = 2019, state = "RI", type = "city", + per_capita = TRUE, adjust_to_year = 2020) + expect_equal(by_id$amt_per_capita_nominal, by_pred$amt_per_capita_nominal) + expect_equal(by_id$amt_per_capita_real, by_pred$amt_per_capita_real) + expect_equal(by_id$pop_source, by_pred$pop_source) + + paged_id <- cog_spending(govid = ids, years = 2019, limit = 5, offset = 5) + paged_pred <- cog_spending(years = 2019, state = "RI", type = "city", + limit = 5, offset = 5) + expect_equal(as.data.frame(paged_id), as.data.frame(paged_pred), + ignore_attr = TRUE) + expect_identical(attr(paged_id, "total_rows"), attr(paged_pred, "total_rows")) + }) +}) + +test_that("a state-only predicate spans every type in that state", { + skip_if_no_corpus() + with_fixture_corpus({ + ids <- cog_gov_search(NULL, state = "DE", type = NULL)$canonical_govid + by_id <- cog_spending(govid = ids, years = 2019) + by_pred <- cog_spending(years = 2019, state = "DE") + expect_equal(nrow(by_id), nrow(by_pred)) + }) +}) + +# --- the postal/FIPS trap -------------------------------------------------- + +test_that("a postal abbreviation resolves to rows, not to silence", { + skip_if_no_corpus() + with_fixture_corpus({ + # canonical_fips_xwalk.fips_state holds "44", not "RI". A predicate built + # from the raw parameter matches nothing and returns an empty result that + # reads as "these governments reported nothing" -- the exact trap cog-api + # hit. A zero-row result here is the regression. + r <- cog_spending(years = 2019, state = "RI", type = "city") + expect_gt(nrow(r), 0L) + + # And the FIPS form is accepted as the same cohort. + expect_equal(nrow(cog_spending(years = 2019, state = "44", type = "city")), + nrow(r)) + }) +}) + +test_that("an unknown state abbreviation aborts with a message that names the problem", { + skip_if_no_corpus() + with_fixture_corpus({ + # Regression: `.state_abbrev_to_fips` is a named character vector, so `[[` + # on an absent name threw base R's "subscript out of bounds" and the + # curated message was unreachable. + expect_error(cog_spending(years = 2019, state = "ZZ"), + class = "uscogdata_unknown_state") + expect_error(cog_spending(years = 2019, state = "ZZ"), + "Unknown state abbreviation") + }) +}) + +# --- naming the cohort ----------------------------------------------------- + +test_that("naming no cohort at all is refused", { + skip_if_no_corpus() + with_fixture_corpus({ + expect_error(cog_spending(years = 2019), class = "uscogdata_no_cohort") + expect_error(cog_revenue(years = 2019), class = "uscogdata_no_cohort") + expect_error(cog_balances(years = 2019), class = "uscogdata_no_cohort") + }) +}) + +test_that("an empty govid vector still fails as it always did", { + skip_if_no_corpus() + with_fixture_corpus({ + # Supplied-but-empty is a caller error, not "cohort named some other way". + expect_error(cog_spending(character(0), 2019), + "must be a non-empty character vector") + }) +}) + +test_that("govid and state/type together intersect", { + skip_if_no_corpus() + with_fixture_corpus({ + cities <- cog_gov_search(NULL, state = "RI", type = "city")$canonical_govid + counties <- cog_gov_search(NULL, state = "DE", type = "county")$canonical_govid + + # The documented rule: the governments in `govid` that ALSO match the + # predicate -- never one silently taking precedence over the other. + both <- cog_spending(govid = c(cities, counties), years = 2019, + state = "RI", type = "city") + only <- cog_spending(govid = cities, years = 2019) + expect_equal(as.data.frame(both), as.data.frame(only), ignore_attr = TRUE) + + # A disjoint intersection is empty, not "whichever one won". + none <- cog_spending(govid = counties, years = 2019, + state = "RI", type = "city") + expect_equal(nrow(none), 0L) + }) +}) + +# --- provenance ------------------------------------------------------------ + +test_that("a govid cohort's provenance scope is untouched", { + skip_if_no_corpus() + with_fixture_corpus({ + ids <- cog_gov_search(NULL, state = "DE", type = "county")$canonical_govid + prov <- attr(cog_spending(govid = ids, years = 2019), "provenance") + expect_setequal(prov$scope$govids_found, ids) + expect_length(prov$scope$govids_missing, 0L) + # No cohort block: govids_found already describes this cohort exactly. + expect_null(prov$scope$cohort) + }) +}) + +test_that("a predicate cohort describes itself instead of listing ids", { + skip_if_no_corpus() + with_fixture_corpus({ + prov <- attr(cog_spending(years = 2019, state = "DE", type = "county"), + "provenance") + # Deliberately NOT the resolved id list: a fleet-scale cohort would put + # 20,000 govids into every response body. + expect_length(prov$scope$govids_found, 0L) + expect_length(prov$scope$govids_missing, 0L) + expect_identical(prov$scope$cohort$state, "DE") + expect_identical(prov$scope$cohort$type, "county") + expect_identical(prov$scope$cohort$n_governments, 3L) + }) +}) + +test_that("the cohort block counts the intersection, not the predicate alone", { + skip_if_no_corpus() + with_fixture_corpus({ + counties <- cog_gov_search(NULL, state = "DE", type = "county")$canonical_govid + prov <- attr( + cog_spending(govid = counties[1], years = 2019, state = "DE", type = "county"), + "provenance" + ) + expect_identical(prov$scope$cohort$n_governments, 1L) + }) +}) + +# --- the SQL actually changed ---------------------------------------------- + +test_that("a predicate cohort never renders the ids into the query", { + skip_if_no_corpus() + with_fixture_corpus({ + ids <- cog_gov_search(NULL, state = "RI", type = "city")$canonical_govid + prov <- attr(cog_spending(years = 2019, state = "RI", type = "city"), + "provenance") + sql <- prov$sql_query %||% prov$sql + skip_if(is.null(sql), "provenance carries no SQL for this verb") + # The whole point: cohort size does not enter the SQL string. + for (id in ids) expect_false(grepl(id, sql, fixed = TRUE)) + expect_match(sql, "SELECT canonical_govid FROM canonical_fips_xwalk", + fixed = TRUE) + }) +}) diff --git a/tests/testthat/test-cohort.R b/tests/testthat/test-cohort.R new file mode 100644 index 0000000..e8c6b0a --- /dev/null +++ b/tests/testthat/test-cohort.R @@ -0,0 +1,83 @@ +# A cohort is how every verb names the set of governments it queries. It can be +# named by explicit id, by a predicate over canonical_fips_xwalk, or by both +# (intersection). These tests cover the SQL construction itself -- pure string +# building, no corpus needed -- because that is where the postal/FIPS trap and +# the 40k-literal blowup both live. + +test_that(".make_cohort() keeps an explicit id vector as ids", { + ch <- .make_cohort(govid = c("550000227544", "060000000001")) + expect_identical(ch$ids, c("550000227544", "060000000001")) + expect_null(ch$state_fips) + expect_null(ch$type_int) + expect_false(.cohort_by_predicate(ch)) +}) + +test_that(".make_cohort() translates a postal abbreviation to FIPS", { + # The trap this whole issue exists to avoid: canonical_fips_xwalk.fips_state + # holds "55", not "WI". A predicate written against the raw parameter matches + # nothing and returns an empty result that reads as "reported nothing". + ch <- .make_cohort(state = "WI") + expect_identical(ch$state_fips, "55") + expect_true(.cohort_by_predicate(ch)) +}) + +test_that(".make_cohort() translates a type label to its integer code", { + expect_identical(.make_cohort(type = "city")$type_int, 2L) + expect_identical(.make_cohort(type = "state")$type_int, 0L) + expect_identical(.make_cohort(type = 1)$type_int, 1L) +}) + +test_that(".make_cohort() reuses the search verb's coercers for invalid input", { + expect_error(.make_cohort(state = "ZZ"), "Unknown state abbreviation") + expect_error(.make_cohort(type = "special_district"), "Unknown type") + # Out-of-scope types (4 = special district, 5 = school district) are refused + # by the numeric branch, with the v0.1-scope message. + expect_error(.make_cohort(type = 4), "type must be 0, 1, 2, or 3") +}) + +test_that(".make_cohort() rejects naming no cohort at all", { + expect_error(.make_cohort(), class = "uscogdata_no_cohort") +}) + +test_that(".cohort_sql() renders an id cohort as a literal IN list", { + sql <- .cohort_sql(.make_cohort(govid = c("a", "b"))) + expect_identical(sql, "canonical_govid IN ('a','b')") +}) + +test_that(".cohort_sql() renders a predicate cohort as an xwalk subquery", { + # The point of the issue: the cohort never becomes a literal list, so its + # size does not enter the SQL string at all. + sql <- .cohort_sql(.make_cohort(state = "WI", type = "city")) + expect_match(sql, "SELECT canonical_govid FROM canonical_fips_xwalk", fixed = TRUE) + expect_match(sql, "fips_state = '55'", fixed = TRUE) + expect_match(sql, "govs_type = 2", fixed = TRUE) + expect_false(grepl("'WI'", sql, fixed = TRUE)) +}) + +test_that(".cohort_sql() renders ids and a predicate as an intersection", { + sql <- .cohort_sql(.make_cohort(govid = c("a", "b"), type = "city")) + expect_match(sql, "canonical_govid IN ('a','b')", fixed = TRUE) + expect_match(sql, "AND canonical_govid IN (SELECT", fixed = TRUE) +}) + +test_that(".cohort_sql() honours a column alias", { + # Several call sites join the xwalk under an alias (`l.`, `x.`, `v.`), so the + # predicate has to be able to name the qualified column. + sql <- .cohort_sql(.make_cohort(state = "WI"), col = "l.canonical_govid") + expect_match(sql, "l.canonical_govid IN (SELECT", fixed = TRUE) + # The subquery's own column stays unqualified -- it selects from the xwalk, + # not from the outer relation. + expect_match(sql, "SELECT canonical_govid FROM", fixed = TRUE) +}) + +test_that(".cohort_sql() escapes quotes in ids", { + sql <- .cohort_sql(.make_cohort(govid = "o'brien")) + expect_match(sql, "'o''brien'", fixed = TRUE) +}) + +test_that("a predicate cohort's SQL does not grow with cohort size", { + # The regression this guards: 20,106 ids rendered to a 301,591-character + # IN list, embedded in 5-8 statements per call. + wide <- .cohort_sql(.make_cohort(state = "CA", type = "city")) + expect_lt(nchar(wide), 200L) +}) diff --git a/tests/testthat/test-recipes.R b/tests/testthat/test-recipes.R index a251119..fdd8d77 100644 --- a/tests/testthat/test-recipes.R +++ b/tests/testthat/test-recipes.R @@ -251,7 +251,7 @@ test_that(".suppressed_components measures the E67/E68 dollars Public Welfare dr s <- uscogdata:::.suppressed_components( con, candidates = c("welfare_cash_e67_wide", "welfare_cash_e68_wide"), - govid = "061037123085", years = 2011L, + cohort = uscogdata:::.make_cohort("061037123085"), years = 2011L, long_view = "spending_long_harmonized", flow_prefixes = c("E", "F", "G")) @@ -269,7 +269,7 @@ test_that(".suppressed_components finds nothing in a modern year", { s <- uscogdata:::.suppressed_components( con, candidates = c("welfare_cash_e67_wide", "welfare_cash_e68_wide"), - govid = "061037123085", years = 2019L, + cohort = uscogdata:::.make_cohort("061037123085"), years = 2019L, long_view = "spending_long_harmonized", flow_prefixes = c("E", "F", "G")) expect_equal(nrow(s), 0L) @@ -280,7 +280,7 @@ test_that(".suppressed_components rejects a long_view outside the allowlist", { con <- uscogdata:::.ensure_session() expect_error( uscogdata:::.suppressed_components( - con, candidates = "welfare_cash_e67_wide", govid = "061037123085", + con, candidates = "welfare_cash_e67_wide", cohort = uscogdata:::.make_cohort("061037123085"), years = 2011L, long_view = "long; DROP TABLE x", flow_prefixes = c("E", "F", "G")), class = "uscogdata_internal_error") @@ -298,7 +298,7 @@ test_that(".suppressed_components never measures a component from the other flow s <- uscogdata:::.suppressed_components( con, candidates = c("welfare_cash_e67_wide", "welfare_cash_e68_wide"), - govid = "061037123085", years = 2011L, + cohort = uscogdata:::.make_cohort("061037123085"), years = 2011L, long_view = "revenue_long_harmonized", flow_prefixes = c("T", "A", "U", "B", "C", "D")) expect_equal(nrow(s), 0L)