feat: cog_recipes + recipe= + signposting suggestions
Adds cog_recipes() to list the curated harmonization_recipes catalog (24 recipes / schema_version >= 5), and a recipe= argument on cog_spending()/ cog_revenue() that runs a recipe's generic multi-code join instead of the category view: SUM(amt * weight) across whichever component codes are present for a (year, canonical_govid), scoped by gov_type_scope. The join deliberately does not filter is_aggregate -- the wide era (<= 2011) exposes these split families (corrections 04+05, IG *89/*47, U4- rents, etc.) ONLY as aggregate rows, with leaf codes first appearing in 2012, so excluding aggregates would zero out the wide-era half of every recipe. This is safe by corpus construction: wide-era rows are aggregate-only, modern rows are leaf-only, and every component is year-scoped, so there is no double-counting. recipe= is mutually exclusive with category=; the result's subtype column reads "recipe" and category reads the recipe's label. Adds recipe-component-driven signposting: when a basis="harmonized" + category query comes back with zero rows in a requested year, and a harmonization recipe covering that category would actually produce rows for this government in that year (via the same join .run_recipe() uses), the recipe is surfaced in provenance$suggestions plus one cli::cli_inform() message. This is deliberately keyed off recipe components rather than harmonization_map's suggested_recipe_id column (which is empty on every live row -- the wide era's split families are NA-by-construction via aggregate exclusion, not an NA ruling to hang a suggestion off of). Also populates the previously-always-empty provenance$series_break_refs (schema v5 only: series_breaks_pq rows whose fin_code is among the observed codes and whose break_year falls in the requested span), and extends cog_explain() with Basis/Harmonization/Recipe/Suggestions/Series breaks sections.
This commit is contained in:
+64
@@ -51,6 +51,15 @@ cog_explain <- function(result, format = c("print", "list")) {
|
||||
cli::cli_text("Category: (all)")
|
||||
}
|
||||
|
||||
if (!is.null(prov$basis)) {
|
||||
note <- if (!is.null(prov$basis_note) && !is.na(prov$basis_note)) {
|
||||
sprintf(" (%s)", prov$basis_note)
|
||||
} else {
|
||||
""
|
||||
}
|
||||
cli::cli_text("Basis: {prov$basis}{note}")
|
||||
}
|
||||
|
||||
cli::cli_h2("Codes observed")
|
||||
codes <- prov$codes_summed$observed
|
||||
if (length(codes) == 0L) {
|
||||
@@ -66,6 +75,39 @@ cog_explain <- function(result, format = c("print", "list")) {
|
||||
)
|
||||
}
|
||||
|
||||
h <- prov$harmonization
|
||||
if (!is.null(h) && isTRUE(h$applied)) {
|
||||
cli::cli_h2("Harmonization")
|
||||
cli::cli_text(
|
||||
"Excluded {h$na_rows_excluded} row(s) with no harmonized_code (${format(h$na_amount_excluded, big.mark = ',')})"
|
||||
)
|
||||
}
|
||||
|
||||
rc <- prov$recipe
|
||||
if (!is.null(rc)) {
|
||||
cli::cli_h2("Recipe")
|
||||
cli::cli_text("{rc$recipe_id}: {rc$label}")
|
||||
comp_lines <- vapply(rc$components, function(x) {
|
||||
sprintf("%s (%s, %s-%s, weight=%s)", x$component_code, x$gov_type_scope,
|
||||
x$year_min, x$year_max, x$weight)
|
||||
}, character(1))
|
||||
cli::cli_ul(comp_lines)
|
||||
}
|
||||
|
||||
if (length(prov$suggestions) > 0L) {
|
||||
cli::cli_h2("Suggestions")
|
||||
sugg_lines <- vapply(prov$suggestions, function(s) {
|
||||
sprintf("%s -- %s (years %s-%s): %s", s$recipe_id, s$label,
|
||||
s$available_years[1], s$available_years[2], s$hint)
|
||||
}, character(1))
|
||||
cli::cli_ul(sugg_lines)
|
||||
}
|
||||
|
||||
if (length(prov$series_break_refs) > 0L) {
|
||||
cli::cli_h2("Series breaks")
|
||||
cli::cli_ul(.series_break_story_lines(prov$series_break_refs))
|
||||
}
|
||||
|
||||
cli::cli_h2("Transformations")
|
||||
uc <- prov$transformations$units_conversion
|
||||
if (isTRUE(uc$applied)) {
|
||||
@@ -109,6 +151,28 @@ cog_explain <- function(result, format = c("print", "list")) {
|
||||
invisible(NULL)
|
||||
}
|
||||
|
||||
# One "break-story" line per referenced break_id: "SB109 (2005): <join_advice>".
|
||||
# Re-queries series_breaks_pq for the detail (break_year, join_advice) that
|
||||
# provenance$series_break_refs deliberately doesn't carry (the schema keeps
|
||||
# that field to a plain id array). Falls back to bare ids if no session is
|
||||
# available (e.g. explaining a result after cog_close()) rather than
|
||||
# erroring cog_explain() over a cosmetic detail.
|
||||
#' @noRd
|
||||
.series_break_story_lines <- function(break_ids) {
|
||||
con <- tryCatch(.ensure_session(), error = function(e) NULL)
|
||||
if (is.null(con) || !DBI::dbIsValid(con)) return(break_ids)
|
||||
detail <- tryCatch(
|
||||
DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT break_id, break_year, join_advice FROM series_breaks_pq
|
||||
WHERE break_id IN (%s) ORDER BY break_id",
|
||||
.sql_lit_chr(break_ids)
|
||||
)),
|
||||
error = function(e) NULL
|
||||
)
|
||||
if (is.null(detail) || nrow(detail) == 0L) return(break_ids)
|
||||
sprintf("%s (%s): %s", detail$break_id, detail$break_year, detail$join_advice)
|
||||
}
|
||||
|
||||
# Expand a 2-digit Census popyear (e.g. 19) to a 4-digit calendar year (2019).
|
||||
# F-33 metadata stores popyear as 2 digits; pivot at 70 to handle a future
|
||||
# corpus that ever spans pre-1970 vintages, though current scope is 2000+.
|
||||
|
||||
+13
-2
@@ -6,7 +6,8 @@
|
||||
per_capita, adjust_to_year, result, sql,
|
||||
subtype_col, basis = NA_character_,
|
||||
basis_note = NA_character_,
|
||||
harmonization = NULL) {
|
||||
harmonization = NULL, recipe = NULL,
|
||||
suggestions = list()) {
|
||||
manifest <- .uscogdata_env$manifest
|
||||
|
||||
codes <- result[["codes_included"]]
|
||||
@@ -31,6 +32,14 @@
|
||||
unique(result$gov_name)
|
||||
}
|
||||
|
||||
schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L))
|
||||
con <- .uscogdata_env$con
|
||||
break_refs <- if (!is.null(con) && DBI::dbIsValid(con)) {
|
||||
.build_series_break_refs(con, codes_observed, years, schema_version)
|
||||
} else {
|
||||
character(0)
|
||||
}
|
||||
|
||||
list(
|
||||
verb = verb,
|
||||
call = paste(deparse(call), collapse = " "),
|
||||
@@ -46,6 +55,8 @@
|
||||
applied = FALSE, na_rows_excluded = 0L, na_amount_excluded = 0,
|
||||
note = NA_character_
|
||||
),
|
||||
recipe = recipe,
|
||||
suggestions = suggestions,
|
||||
scope = list(
|
||||
gov_types_included = as.integer(unlist(manifest$scope$gov_types_included)),
|
||||
gov_types_excluded = as.integer(unlist(manifest$scope$gov_types_excluded)),
|
||||
@@ -98,7 +109,7 @@
|
||||
index = if (is.null(adjust_to_year)) NA_character_ else "CPI-U (BLS CPIAUCSL annual average, bundled)"
|
||||
)
|
||||
),
|
||||
series_break_refs = character(0),
|
||||
series_break_refs = break_refs,
|
||||
manifest = list(
|
||||
schema_version = as.integer(manifest$schema_version),
|
||||
pipeline_commit = manifest$pipeline_commit %||% NA_character_,
|
||||
|
||||
+167
@@ -0,0 +1,167 @@
|
||||
# R/recipes.R
|
||||
# Harmonization recipes: multi-code, cross-vintage series built by summing a
|
||||
# fixed set of component item codes with per-component weights and
|
||||
# year/gov-type scoping (see the `harmonization_recipes` view, registered
|
||||
# from data/harmonization_recipes.parquet, schema_version >= 5 only).
|
||||
#
|
||||
# Recipes exist because some cross-vintage series can't be expressed as a
|
||||
# 1:1 harmonized_code mapping (basis = "harmonized"): the wide era (pre-2012)
|
||||
# publishes only a combined aggregate row for these families (e.g.
|
||||
# corrections functions 04+05), while the modern era splits them into leaf
|
||||
# codes. A recipe's generic join sums whichever of its component codes are
|
||||
# present for a given year, so the resulting series is continuous across
|
||||
# that format boundary.
|
||||
|
||||
#' List available harmonization recipes
|
||||
#'
|
||||
#' Recipes are multi-code cross-vintage series (see [cog_spending()]'s
|
||||
#' `recipe` argument) catalogued in the corpus's `harmonization_recipes`
|
||||
#' table. Use this to discover valid `recipe` ids.
|
||||
#'
|
||||
#' @param pattern Optional regex matched case-insensitively against
|
||||
#' `recipe_id` or `label`.
|
||||
#' @return Tibble with columns `recipe_id`, `label`, `n_components`,
|
||||
#' `year_min`, `year_max` (the min/max component year coverage), sorted by
|
||||
#' `recipe_id`.
|
||||
#' @export
|
||||
cog_recipes <- function(pattern = NULL) {
|
||||
if (!is.null(pattern) &&
|
||||
(!is.character(pattern) || length(pattern) != 1L)) {
|
||||
cli::cli_abort("`pattern` must be a length-1 character string or NULL.")
|
||||
}
|
||||
con <- .ensure_session()
|
||||
.require_schema_v5(con, .uscogdata_env$manifest, "cog_recipes()")
|
||||
|
||||
where <- if (is.null(pattern)) {
|
||||
""
|
||||
} else {
|
||||
sprintf(
|
||||
"WHERE regexp_matches(recipe_id, %1$s, 'i') OR regexp_matches(label, %1$s, 'i')",
|
||||
.sql_lit_chr(pattern)
|
||||
)
|
||||
}
|
||||
sql <- paste(
|
||||
"SELECT recipe_id, any_value(label) AS label,
|
||||
COUNT(*) AS n_components,
|
||||
MIN(year_min) AS year_min, MAX(year_max) AS year_max
|
||||
FROM harmonization_recipes",
|
||||
where,
|
||||
"GROUP BY recipe_id
|
||||
ORDER BY recipe_id"
|
||||
)
|
||||
out <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
out$year_min <- as.integer(out$year_min)
|
||||
out$year_max <- as.integer(out$year_max)
|
||||
out$n_components <- as.integer(out$n_components)
|
||||
out
|
||||
}
|
||||
|
||||
#' Abort unless the active corpus has schema_version >= 5.
|
||||
#' @noRd
|
||||
.require_schema_v5 <- function(con, manifest, what) {
|
||||
sv <- suppressWarnings(as.integer(manifest$schema_version %||% 0L))
|
||||
if (sv < 5L) {
|
||||
cli::cli_abort(c(
|
||||
sprintf("%s requires corpus schema_version >= 5.", what),
|
||||
x = "Active corpus has schema_version {sv}.",
|
||||
i = "Point USCOGDATA_URL at a schema_version >= 5 corpus to use harmonization recipes."
|
||||
), class = "uscogdata_schema_unsupported")
|
||||
}
|
||||
invisible(sv)
|
||||
}
|
||||
|
||||
#' Abort with the valid id list unless `recipe_id` exists in the catalog.
|
||||
#' @noRd
|
||||
.validate_recipe_id <- function(con, recipe_id) {
|
||||
ids <- DBI::dbGetQuery(
|
||||
con, "SELECT DISTINCT recipe_id FROM harmonization_recipes"
|
||||
)$recipe_id
|
||||
if (!recipe_id %in% ids) {
|
||||
cli::cli_abort(c(
|
||||
"Unknown recipe = {.val {recipe_id}}.",
|
||||
i = "Valid ids: {paste(sort(ids), collapse = ', ')}",
|
||||
i = "See cog_recipes() for labels and year coverage."
|
||||
), class = "uscogdata_unknown_recipe")
|
||||
}
|
||||
invisible(TRUE)
|
||||
}
|
||||
|
||||
#' Fetch the component rows for one recipe (label, component codes, scope,
|
||||
#' year ranges, weights) -- both for running the recipe and for the
|
||||
#' `recipe` provenance block.
|
||||
#' @noRd
|
||||
.recipe_components <- function(con, recipe_id) {
|
||||
sql <- sprintf(
|
||||
"SELECT recipe_id, label, component_code, gov_type_scope,
|
||||
year_min, year_max, weight, source_break_ids, notes
|
||||
FROM harmonization_recipes
|
||||
WHERE recipe_id = %s
|
||||
ORDER BY component_code",
|
||||
.sql_lit_chr(recipe_id)
|
||||
)
|
||||
tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
}
|
||||
|
||||
#' Run a recipe's generic join: sum `amt * weight` across whichever
|
||||
#' component codes are present for each (year, canonical_govid), scoped by
|
||||
#' gov_type_scope. Deliberately does NOT filter `NOT is_aggregate`: in the
|
||||
#' wide era (<= 2011) these families' component codes exist ONLY as
|
||||
#' aggregate rows (leaves first appear 2012), so excluding aggregates would
|
||||
#' zero out the wide-era half of every recipe. This is safe by corpus
|
||||
#' construction -- wide-era rows for these codes are aggregate-only, modern
|
||||
#' rows are leaf-only, and every component row is year-scoped via
|
||||
#' `year_min`/`year_max` -- so there is no double-counting. (Checkpoint
|
||||
#' review docs/phase_r_harmonization_review.md § 0.2.)
|
||||
#' @noRd
|
||||
.run_recipe <- function(con, recipe_id, govid, years) {
|
||||
sql <- sprintf(
|
||||
"SELECT l.year, l.canonical_govid,
|
||||
COALESCE(x.gov_name, l.gov_name) AS gov_name,
|
||||
SUM(l.amt * r.weight) * 1000.0 AS amt_nominal,
|
||||
string_agg(DISTINCT l.item_code, ',' ORDER BY l.item_code) AS codes_included
|
||||
FROM long l
|
||||
JOIN harmonization_recipes r
|
||||
ON l.item_code = r.component_code
|
||||
AND l.year BETWEEN r.year_min AND r.year_max
|
||||
AND (r.gov_type_scope = 'all'
|
||||
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
|
||||
WHERE r.recipe_id = %1$s
|
||||
AND l.canonical_govid IN (%2$s)
|
||||
AND l.year IN (%3$s)
|
||||
GROUP BY 1, 2, 3
|
||||
ORDER BY 1, 2",
|
||||
.sql_lit_chr(recipe_id), .sql_lit_chr(govid),
|
||||
paste(as.integer(years), collapse = ",")
|
||||
)
|
||||
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
attr(result, "sql_query") <- sql
|
||||
result
|
||||
}
|
||||
|
||||
#' Shape a raw .run_recipe() result into the standard cog_spending()/
|
||||
#' cog_revenue() column layout: subtype = "recipe", category = the recipe's
|
||||
#' label, aggregate_fallback = FALSE (recipes resolve coverage gaps by
|
||||
#' construction, not by falling back to an aggregate row).
|
||||
#' @noRd
|
||||
.shape_recipe_result <- function(result, subtype_col, label) {
|
||||
sql_query <- attr(result, "sql_query")
|
||||
n <- nrow(result)
|
||||
result[[subtype_col]] <- rep("recipe", n)
|
||||
result$category <- rep(label, n)
|
||||
result$aggregate_fallback <- rep(FALSE, n)
|
||||
result <- result[, c(
|
||||
"year", "canonical_govid", "gov_name", subtype_col, "category",
|
||||
"amt_nominal", "codes_included", "aggregate_fallback"
|
||||
), drop = FALSE]
|
||||
attr(result, "sql_query") <- sql_query
|
||||
result
|
||||
}
|
||||
|
||||
#' Turn a small data.frame into a list-of-lists (one list per row), the
|
||||
#' shape used for the `recipe$components` provenance block.
|
||||
#' @noRd
|
||||
.df_to_row_list <- function(df) {
|
||||
lapply(seq_len(nrow(df)), function(i) as.list(df[i, , drop = FALSE]))
|
||||
}
|
||||
+3
-2
@@ -15,7 +15,7 @@
|
||||
#' @export
|
||||
cog_revenue <- function(govid, years, category = NULL,
|
||||
per_capita = FALSE, adjust_to_year = NULL,
|
||||
basis = c("harmonized", "raw")) {
|
||||
basis = c("harmonized", "raw"), recipe = NULL) {
|
||||
.verb_spendrev(
|
||||
verb = "cog_revenue",
|
||||
view_base = "revenue_annotated",
|
||||
@@ -27,6 +27,7 @@ cog_revenue <- function(govid, years, category = NULL,
|
||||
category = category,
|
||||
per_capita = per_capita,
|
||||
adjust_to_year = adjust_to_year,
|
||||
basis = basis
|
||||
basis = basis,
|
||||
recipe = recipe
|
||||
)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
# R/series_breaks.R
|
||||
# Populates prov$series_break_refs (schema in inst/schemas/provenance-v1.json
|
||||
# defines the field; it was always present but always empty pre-Phase-R2)
|
||||
# with the ids of any catalogued series break whose fin_code appears among
|
||||
# the result's observed item codes and whose break_year falls inside the
|
||||
# requested year span -- the "break warnings in the provenance envelope"
|
||||
# spec § 5 promises downstream consumers (cog-api passes provenance through
|
||||
# verbatim). schema_version >= 5 only: series_breaks_pq isn't registered on
|
||||
# an older corpus.
|
||||
|
||||
#' @noRd
|
||||
.build_series_break_refs <- function(con, codes_observed, years, schema_version) {
|
||||
if (schema_version < 5L || length(codes_observed) == 0L) return(character(0))
|
||||
sql <- sprintf(
|
||||
"SELECT DISTINCT break_id
|
||||
FROM series_breaks_pq
|
||||
WHERE fin_code IN (%s) AND break_year BETWEEN %d AND %d
|
||||
ORDER BY break_id",
|
||||
.sql_lit_chr(codes_observed), min(as.integer(years)), max(as.integer(years))
|
||||
)
|
||||
DBI::dbGetQuery(con, sql)$break_id
|
||||
}
|
||||
+57
-10
@@ -29,6 +29,12 @@
|
||||
#' tables), `basis` silently resolves to `"raw"` when left at its default
|
||||
#' and the resolution is recorded in the provenance; explicitly passing
|
||||
#' `basis = "harmonized"` on such a corpus aborts.
|
||||
#' @param recipe Optional harmonization recipe id (see [cog_recipes()]) for
|
||||
#' multi-code cross-vintage series that a 1:1 harmonized_code mapping
|
||||
#' can't express (e.g. a wide-era aggregate that only splits into leaf
|
||||
#' codes in the modern era). Mutually exclusive with `category`. The
|
||||
#' result's subtype column reads `"recipe"` and `category` reads the
|
||||
#' recipe's label. Requires `schema_version >= 5`.
|
||||
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||
#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
||||
#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
||||
@@ -37,7 +43,7 @@
|
||||
#' @export
|
||||
cog_spending <- function(govid, years, category = NULL,
|
||||
per_capita = FALSE, adjust_to_year = NULL,
|
||||
basis = c("harmonized", "raw")) {
|
||||
basis = c("harmonized", "raw"), recipe = NULL) {
|
||||
.verb_spendrev(
|
||||
verb = "cog_spending",
|
||||
view_base = "spending_annotated",
|
||||
@@ -49,7 +55,8 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
category = category,
|
||||
per_capita = per_capita,
|
||||
adjust_to_year = adjust_to_year,
|
||||
basis = basis
|
||||
basis = basis,
|
||||
recipe = recipe
|
||||
)
|
||||
}
|
||||
|
||||
@@ -57,12 +64,13 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
.verb_spendrev <- function(verb, view_base, subtype_col, flow_prefixes, call,
|
||||
govid, years, category,
|
||||
per_capita, adjust_to_year,
|
||||
basis = c("harmonized", "raw")) {
|
||||
basis = c("harmonized", "raw"), recipe = NULL) {
|
||||
basis_explicit <- length(basis) == 1L
|
||||
basis <- match.arg(basis, c("harmonized", "raw"))
|
||||
|
||||
govid <- .coerce_govid_input(govid, arg = "govid")
|
||||
.validate_verb_inputs(govid, years, category, per_capita, adjust_to_year)
|
||||
.validate_verb_inputs(govid, years, category, per_capita, adjust_to_year,
|
||||
recipe)
|
||||
|
||||
years <- as.integer(years)
|
||||
if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year)
|
||||
@@ -73,9 +81,26 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
|
||||
resolved <- .resolve_basis(basis, basis_explicit, manifest)
|
||||
|
||||
view <- .select_view(view_base, resolved$basis)
|
||||
sql <- .build_verb_sql(view, subtype_col, govid, years, category)
|
||||
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
recipe_block <- NULL
|
||||
category_for_prov <- category
|
||||
if (!is.null(recipe)) {
|
||||
.require_schema_v5(con, manifest, "recipe =")
|
||||
.validate_recipe_id(con, recipe)
|
||||
comps <- .recipe_components(con, recipe)
|
||||
recipe_label <- comps$label[[1]]
|
||||
result <- .run_recipe(con, recipe, govid, years)
|
||||
sql <- attr(result, "sql_query")
|
||||
result <- .shape_recipe_result(result, subtype_col, recipe_label)
|
||||
recipe_block <- list(
|
||||
recipe_id = recipe, label = recipe_label,
|
||||
components = .df_to_row_list(comps)
|
||||
)
|
||||
category_for_prov <- recipe_label
|
||||
} else {
|
||||
view <- .select_view(view_base, resolved$basis)
|
||||
sql <- .build_verb_sql(view, subtype_col, govid, years, category)
|
||||
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
}
|
||||
|
||||
if (per_capita) result <- .attach_per_capita(result, con, govid)
|
||||
if (!is.null(adjust_to_year)) {
|
||||
@@ -88,12 +113,18 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
con, govid, years, resolved, flow_prefixes
|
||||
)
|
||||
|
||||
suggestions <- if (is.null(recipe)) {
|
||||
.build_suggestions(con, govid, years, category, result, resolved$basis)
|
||||
} else {
|
||||
list()
|
||||
}
|
||||
|
||||
prov <- .build_provenance(
|
||||
verb = verb,
|
||||
call = call,
|
||||
govid = govid,
|
||||
years = years,
|
||||
category = category,
|
||||
category = category_for_prov,
|
||||
per_capita = per_capita,
|
||||
adjust_to_year = adjust_to_year,
|
||||
result = result,
|
||||
@@ -101,18 +132,23 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
subtype_col = subtype_col,
|
||||
basis = resolved$basis,
|
||||
basis_note = resolved$note,
|
||||
harmonization = harmonization
|
||||
harmonization = harmonization,
|
||||
recipe = recipe_block,
|
||||
suggestions = suggestions
|
||||
)
|
||||
prov$scope$govids_found <- scope$found
|
||||
prov$scope$govids_missing <- scope$missing
|
||||
attr(result, "provenance") <- prov
|
||||
attr(result, ".popyear_range") <- NULL
|
||||
|
||||
if (length(suggestions) > 0L) .inform_suggestions(suggestions)
|
||||
|
||||
result
|
||||
}
|
||||
|
||||
#' @noRd
|
||||
.validate_verb_inputs <- function(govid, years, category,
|
||||
per_capita, adjust_to_year) {
|
||||
per_capita, adjust_to_year, recipe = NULL) {
|
||||
if (!is.character(govid) || length(govid) == 0L) {
|
||||
cli::cli_abort("`govid` must be a non-empty character vector.")
|
||||
}
|
||||
@@ -131,6 +167,17 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
cli::cli_abort("`adjust_to_year` must be NULL or a length-1 integer.")
|
||||
}
|
||||
}
|
||||
if (!is.null(recipe)) {
|
||||
if (!is.character(recipe) || length(recipe) != 1L) {
|
||||
cli::cli_abort("`recipe` must be NULL or a length-1 character string.")
|
||||
}
|
||||
if (!is.null(category)) {
|
||||
cli::cli_abort(c(
|
||||
"`recipe` and `category` are mutually exclusive.",
|
||||
i = "Pass one or the other, not both."
|
||||
), class = "uscogdata_recipe_category_conflict")
|
||||
}
|
||||
}
|
||||
invisible(TRUE)
|
||||
}
|
||||
|
||||
|
||||
+119
@@ -0,0 +1,119 @@
|
||||
# R/suggestions.R
|
||||
# Recipe-component-driven signposting: when a basis = "harmonized" query for
|
||||
# a category comes back with a coverage gap in some requested years (the
|
||||
# result has no rows at all in that year) that a harmonization recipe would
|
||||
# actually fill for this government, surface that recipe as a suggestion.
|
||||
#
|
||||
# This is deliberately keyed off the recipe catalog's component codes, not
|
||||
# off harmonization_map rows: no live map row carries a non-blank
|
||||
# suggested_recipe_id (the corpus's wide era exposes split families like
|
||||
# corrections functions 04+05 ONLY as aggregate rows, which basis =
|
||||
# "harmonized" excludes by construction -- there's no NA ruling to hang a
|
||||
# suggestion off of, just a leaf-code absence a recipe happens to fill).
|
||||
# See docs/phase_r_harmonization_review.md § 0.3.
|
||||
#
|
||||
# Scope is deliberately narrow: signposting only runs when the caller
|
||||
# supplied a `category` (an un-scoped, all-categories query has no single
|
||||
# coverage question to answer) and only flags a recipe when the ACTUAL
|
||||
# result has zero rows in a requested year AND the candidate recipe's own
|
||||
# generic join (same join .run_recipe() uses, including its wide-era
|
||||
# aggregate rows) produces at least one row for this government in that
|
||||
# year. Checking presence per-government (not corpus-wide) avoids false
|
||||
# positives from ordinary reporting variance -- most governments don't use
|
||||
# every sibling code in a multi-code category every year, and that is not
|
||||
# a format-boundary gap worth signposting.
|
||||
|
||||
#' Build the `prov$suggestions` list for a (non-recipe) basis = "harmonized"
|
||||
#' verb call: recipes whose generic join would fill a real gap in `result`.
|
||||
#'
|
||||
#' @param con Active DuckDB connection.
|
||||
#' @param govid Character vector of canonical_govid values (the verb's raw
|
||||
#' `govid`).
|
||||
#' @param years Integer vector of requested years.
|
||||
#' @param category `category` argument as passed to the verb (character
|
||||
#' vector or `NULL`; suggestions are only computed when non-NULL).
|
||||
#' @param result The verb's already-computed result tibble (post basis
|
||||
#' query, pre per_capita/adjust_to_year).
|
||||
#' @param basis The *resolved* basis (`"harmonized"` or `"raw"`).
|
||||
#' @return List of `list(recipe_id, label, available_years, hint)`, possibly
|
||||
#' empty.
|
||||
#' @noRd
|
||||
.build_suggestions <- function(con, govid, years, category, result, basis) {
|
||||
if (!identical(basis, "harmonized") || is.null(category)) return(list())
|
||||
|
||||
candidates <- DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||
WHERE component_code IN (
|
||||
SELECT DISTINCT item_code FROM summary_categories WHERE category IN (%s)
|
||||
)",
|
||||
.sql_lit_chr(category)
|
||||
))$recipe_id
|
||||
if (length(candidates) == 0L) return(list())
|
||||
|
||||
result_years <- if (is.null(result) || nrow(result) == 0L) {
|
||||
integer(0)
|
||||
} else {
|
||||
unique(as.integer(result$year))
|
||||
}
|
||||
gap_years <- setdiff(as.integer(years), result_years)
|
||||
if (length(gap_years) == 0L) return(list())
|
||||
|
||||
meta <- tibble::as_tibble(DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT recipe_id, any_value(label) AS label,
|
||||
MIN(year_min) AS year_min, MAX(year_max) AS year_max
|
||||
FROM harmonization_recipes
|
||||
WHERE recipe_id IN (%s)
|
||||
GROUP BY recipe_id",
|
||||
.sql_lit_chr(candidates)
|
||||
)))
|
||||
|
||||
# Which (recipe_id, year) pairs the recipe's own generic join actually
|
||||
# covers for this government, restricted to the gap years -- the same
|
||||
# join .run_recipe() uses (component year_min/year_max + gov_type_scope,
|
||||
# no is_aggregate filter), just checking existence instead of summing.
|
||||
covered <- DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT DISTINCT r.recipe_id, l.year
|
||||
FROM long l
|
||||
JOIN harmonization_recipes r
|
||||
ON l.item_code = r.component_code
|
||||
AND l.year BETWEEN r.year_min AND r.year_max
|
||||
AND (r.gov_type_scope = 'all'
|
||||
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||
WHERE r.recipe_id IN (%s)
|
||||
AND l.canonical_govid IN (%s)
|
||||
AND l.year IN (%s)",
|
||||
.sql_lit_chr(candidates), .sql_lit_chr(govid),
|
||||
paste(gap_years, collapse = ",")
|
||||
))
|
||||
|
||||
suggestions <- list()
|
||||
for (rid in candidates) {
|
||||
if (!rid %in% covered$recipe_id) next
|
||||
m <- meta[meta$recipe_id == rid, ]
|
||||
suggestions[[length(suggestions) + 1L]] <- list(
|
||||
recipe_id = rid,
|
||||
label = m$label[[1]],
|
||||
available_years = c(as.integer(m$year_min), as.integer(m$year_max)),
|
||||
hint = sprintf("re-run with recipe = '%s'", rid)
|
||||
)
|
||||
}
|
||||
suggestions
|
||||
}
|
||||
|
||||
#' Emit the single cli::cli_inform() message summarizing all suggestions
|
||||
#' for a verb call (the brief's "one message", not one per suggestion).
|
||||
#' Bullet text is pre-formatted plain text (no cli/glue `{}` markup) since
|
||||
#' recipe ids/labels are untrusted-ish data values, not literal call-site
|
||||
#' expressions.
|
||||
#' @noRd
|
||||
.inform_suggestions <- function(suggestions) {
|
||||
bullets <- vapply(suggestions, function(s) {
|
||||
sprintf("%s (%d-%d): %s", s$recipe_id,
|
||||
s$available_years[1], s$available_years[2], s$hint)
|
||||
}, character(1))
|
||||
cli::cli_inform(c(
|
||||
i = "Coverage gap detected for the requested years; a harmonization recipe may fill it:",
|
||||
stats::setNames(bullets, rep("*", length(bullets)))
|
||||
))
|
||||
}
|
||||
Reference in New Issue
Block a user