From 4de915b557f684040b940e9668662164b461a497 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Sat, 18 Jul 2026 23:33:02 -0400 Subject: [PATCH] feat: cog_recipes + recipe= + signposting suggestions Adds cog_recipes() to list the curated harmonization_recipes catalog (24 recipes / schema_version >= 5), and a recipe= argument on cog_spending()/ cog_revenue() that runs a recipe's generic multi-code join instead of the category view: SUM(amt * weight) across whichever component codes are present for a (year, canonical_govid), scoped by gov_type_scope. The join deliberately does not filter is_aggregate -- the wide era (<= 2011) exposes these split families (corrections 04+05, IG *89/*47, U4- rents, etc.) ONLY as aggregate rows, with leaf codes first appearing in 2012, so excluding aggregates would zero out the wide-era half of every recipe. This is safe by corpus construction: wide-era rows are aggregate-only, modern rows are leaf-only, and every component is year-scoped, so there is no double-counting. recipe= is mutually exclusive with category=; the result's subtype column reads "recipe" and category reads the recipe's label. Adds recipe-component-driven signposting: when a basis="harmonized" + category query comes back with zero rows in a requested year, and a harmonization recipe covering that category would actually produce rows for this government in that year (via the same join .run_recipe() uses), the recipe is surfaced in provenance$suggestions plus one cli::cli_inform() message. This is deliberately keyed off recipe components rather than harmonization_map's suggested_recipe_id column (which is empty on every live row -- the wide era's split families are NA-by-construction via aggregate exclusion, not an NA ruling to hang a suggestion off of). Also populates the previously-always-empty provenance$series_break_refs (schema v5 only: series_breaks_pq rows whose fin_code is among the observed codes and whose break_year falls in the requested span), and extends cog_explain() with Basis/Harmonization/Recipe/Suggestions/Series breaks sections. --- NAMESPACE | 1 + R/explain.R | 64 +++++++++++ R/provenance.R | 15 ++- R/recipes.R | 167 +++++++++++++++++++++++++++++ R/revenue.R | 5 +- R/series_breaks.R | 22 ++++ R/spending.R | 67 ++++++++++-- R/suggestions.R | 119 +++++++++++++++++++++ man/cog_recipes.Rd | 22 ++++ man/cog_revenue.Rd | 10 +- man/cog_spending.Rd | 10 +- tests/testthat/test-explain.R | 39 +++++++ tests/testthat/test-recipes.R | 188 +++++++++++++++++++++++++++++++++ tests/testthat/test-spending.R | 15 +++ tests/testthat/test-views.R | 23 ++++ 15 files changed, 751 insertions(+), 16 deletions(-) create mode 100644 R/recipes.R create mode 100644 R/series_breaks.R create mode 100644 R/suggestions.R create mode 100644 man/cog_recipes.Rd create mode 100644 tests/testthat/test-recipes.R diff --git a/NAMESPACE b/NAMESPACE index 19baa13..1c67f0b 100644 --- a/NAMESPACE +++ b/NAMESPACE @@ -10,5 +10,6 @@ export(cog_gov_search) export(cog_manifest) export(cog_mirror) export(cog_peer_compare) +export(cog_recipes) export(cog_revenue) export(cog_spending) diff --git a/R/explain.R b/R/explain.R index 0cec305..2f2d44b 100644 --- a/R/explain.R +++ b/R/explain.R @@ -51,6 +51,15 @@ cog_explain <- function(result, format = c("print", "list")) { cli::cli_text("Category: (all)") } + if (!is.null(prov$basis)) { + note <- if (!is.null(prov$basis_note) && !is.na(prov$basis_note)) { + sprintf(" (%s)", prov$basis_note) + } else { + "" + } + cli::cli_text("Basis: {prov$basis}{note}") + } + cli::cli_h2("Codes observed") codes <- prov$codes_summed$observed if (length(codes) == 0L) { @@ -66,6 +75,39 @@ cog_explain <- function(result, format = c("print", "list")) { ) } + h <- prov$harmonization + if (!is.null(h) && isTRUE(h$applied)) { + cli::cli_h2("Harmonization") + cli::cli_text( + "Excluded {h$na_rows_excluded} row(s) with no harmonized_code (${format(h$na_amount_excluded, big.mark = ',')})" + ) + } + + rc <- prov$recipe + if (!is.null(rc)) { + cli::cli_h2("Recipe") + cli::cli_text("{rc$recipe_id}: {rc$label}") + comp_lines <- vapply(rc$components, function(x) { + sprintf("%s (%s, %s-%s, weight=%s)", x$component_code, x$gov_type_scope, + x$year_min, x$year_max, x$weight) + }, character(1)) + cli::cli_ul(comp_lines) + } + + if (length(prov$suggestions) > 0L) { + cli::cli_h2("Suggestions") + sugg_lines <- vapply(prov$suggestions, function(s) { + sprintf("%s -- %s (years %s-%s): %s", s$recipe_id, s$label, + s$available_years[1], s$available_years[2], s$hint) + }, character(1)) + cli::cli_ul(sugg_lines) + } + + if (length(prov$series_break_refs) > 0L) { + cli::cli_h2("Series breaks") + cli::cli_ul(.series_break_story_lines(prov$series_break_refs)) + } + cli::cli_h2("Transformations") uc <- prov$transformations$units_conversion if (isTRUE(uc$applied)) { @@ -109,6 +151,28 @@ cog_explain <- function(result, format = c("print", "list")) { invisible(NULL) } +# One "break-story" line per referenced break_id: "SB109 (2005): ". +# Re-queries series_breaks_pq for the detail (break_year, join_advice) that +# provenance$series_break_refs deliberately doesn't carry (the schema keeps +# that field to a plain id array). Falls back to bare ids if no session is +# available (e.g. explaining a result after cog_close()) rather than +# erroring cog_explain() over a cosmetic detail. +#' @noRd +.series_break_story_lines <- function(break_ids) { + con <- tryCatch(.ensure_session(), error = function(e) NULL) + if (is.null(con) || !DBI::dbIsValid(con)) return(break_ids) + detail <- tryCatch( + DBI::dbGetQuery(con, sprintf( + "SELECT break_id, break_year, join_advice FROM series_breaks_pq + WHERE break_id IN (%s) ORDER BY break_id", + .sql_lit_chr(break_ids) + )), + error = function(e) NULL + ) + if (is.null(detail) || nrow(detail) == 0L) return(break_ids) + sprintf("%s (%s): %s", detail$break_id, detail$break_year, detail$join_advice) +} + # Expand a 2-digit Census popyear (e.g. 19) to a 4-digit calendar year (2019). # F-33 metadata stores popyear as 2 digits; pivot at 70 to handle a future # corpus that ever spans pre-1970 vintages, though current scope is 2000+. diff --git a/R/provenance.R b/R/provenance.R index fbd3cb0..bfd7a03 100644 --- a/R/provenance.R +++ b/R/provenance.R @@ -6,7 +6,8 @@ per_capita, adjust_to_year, result, sql, subtype_col, basis = NA_character_, basis_note = NA_character_, - harmonization = NULL) { + harmonization = NULL, recipe = NULL, + suggestions = list()) { manifest <- .uscogdata_env$manifest codes <- result[["codes_included"]] @@ -31,6 +32,14 @@ unique(result$gov_name) } + schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L)) + con <- .uscogdata_env$con + break_refs <- if (!is.null(con) && DBI::dbIsValid(con)) { + .build_series_break_refs(con, codes_observed, years, schema_version) + } else { + character(0) + } + list( verb = verb, call = paste(deparse(call), collapse = " "), @@ -46,6 +55,8 @@ applied = FALSE, na_rows_excluded = 0L, na_amount_excluded = 0, note = NA_character_ ), + recipe = recipe, + suggestions = suggestions, scope = list( gov_types_included = as.integer(unlist(manifest$scope$gov_types_included)), gov_types_excluded = as.integer(unlist(manifest$scope$gov_types_excluded)), @@ -98,7 +109,7 @@ index = if (is.null(adjust_to_year)) NA_character_ else "CPI-U (BLS CPIAUCSL annual average, bundled)" ) ), - series_break_refs = character(0), + series_break_refs = break_refs, manifest = list( schema_version = as.integer(manifest$schema_version), pipeline_commit = manifest$pipeline_commit %||% NA_character_, diff --git a/R/recipes.R b/R/recipes.R new file mode 100644 index 0000000..11a55be --- /dev/null +++ b/R/recipes.R @@ -0,0 +1,167 @@ +# R/recipes.R +# Harmonization recipes: multi-code, cross-vintage series built by summing a +# fixed set of component item codes with per-component weights and +# year/gov-type scoping (see the `harmonization_recipes` view, registered +# from data/harmonization_recipes.parquet, schema_version >= 5 only). +# +# Recipes exist because some cross-vintage series can't be expressed as a +# 1:1 harmonized_code mapping (basis = "harmonized"): the wide era (pre-2012) +# publishes only a combined aggregate row for these families (e.g. +# corrections functions 04+05), while the modern era splits them into leaf +# codes. A recipe's generic join sums whichever of its component codes are +# present for a given year, so the resulting series is continuous across +# that format boundary. + +#' List available harmonization recipes +#' +#' Recipes are multi-code cross-vintage series (see [cog_spending()]'s +#' `recipe` argument) catalogued in the corpus's `harmonization_recipes` +#' table. Use this to discover valid `recipe` ids. +#' +#' @param pattern Optional regex matched case-insensitively against +#' `recipe_id` or `label`. +#' @return Tibble with columns `recipe_id`, `label`, `n_components`, +#' `year_min`, `year_max` (the min/max component year coverage), sorted by +#' `recipe_id`. +#' @export +cog_recipes <- function(pattern = NULL) { + if (!is.null(pattern) && + (!is.character(pattern) || length(pattern) != 1L)) { + cli::cli_abort("`pattern` must be a length-1 character string or NULL.") + } + con <- .ensure_session() + .require_schema_v5(con, .uscogdata_env$manifest, "cog_recipes()") + + where <- if (is.null(pattern)) { + "" + } else { + sprintf( + "WHERE regexp_matches(recipe_id, %1$s, 'i') OR regexp_matches(label, %1$s, 'i')", + .sql_lit_chr(pattern) + ) + } + sql <- paste( + "SELECT recipe_id, any_value(label) AS label, + COUNT(*) AS n_components, + MIN(year_min) AS year_min, MAX(year_max) AS year_max + FROM harmonization_recipes", + where, + "GROUP BY recipe_id + ORDER BY recipe_id" + ) + out <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) + out$year_min <- as.integer(out$year_min) + out$year_max <- as.integer(out$year_max) + out$n_components <- as.integer(out$n_components) + out +} + +#' Abort unless the active corpus has schema_version >= 5. +#' @noRd +.require_schema_v5 <- function(con, manifest, what) { + sv <- suppressWarnings(as.integer(manifest$schema_version %||% 0L)) + if (sv < 5L) { + cli::cli_abort(c( + sprintf("%s requires corpus schema_version >= 5.", what), + x = "Active corpus has schema_version {sv}.", + i = "Point USCOGDATA_URL at a schema_version >= 5 corpus to use harmonization recipes." + ), class = "uscogdata_schema_unsupported") + } + invisible(sv) +} + +#' Abort with the valid id list unless `recipe_id` exists in the catalog. +#' @noRd +.validate_recipe_id <- function(con, recipe_id) { + ids <- DBI::dbGetQuery( + con, "SELECT DISTINCT recipe_id FROM harmonization_recipes" + )$recipe_id + if (!recipe_id %in% ids) { + cli::cli_abort(c( + "Unknown recipe = {.val {recipe_id}}.", + i = "Valid ids: {paste(sort(ids), collapse = ', ')}", + i = "See cog_recipes() for labels and year coverage." + ), class = "uscogdata_unknown_recipe") + } + invisible(TRUE) +} + +#' Fetch the component rows for one recipe (label, component codes, scope, +#' year ranges, weights) -- both for running the recipe and for the +#' `recipe` provenance block. +#' @noRd +.recipe_components <- function(con, recipe_id) { + sql <- sprintf( + "SELECT recipe_id, label, component_code, gov_type_scope, + year_min, year_max, weight, source_break_ids, notes + FROM harmonization_recipes + WHERE recipe_id = %s + ORDER BY component_code", + .sql_lit_chr(recipe_id) + ) + tibble::as_tibble(DBI::dbGetQuery(con, sql)) +} + +#' Run a recipe's generic join: sum `amt * weight` across whichever +#' component codes are present for each (year, canonical_govid), scoped by +#' gov_type_scope. Deliberately does NOT filter `NOT is_aggregate`: in the +#' wide era (<= 2011) these families' component codes exist ONLY as +#' aggregate rows (leaves first appear 2012), so excluding aggregates would +#' zero out the wide-era half of every recipe. This is safe by corpus +#' construction -- wide-era rows for these codes are aggregate-only, modern +#' rows are leaf-only, and every component row is year-scoped via +#' `year_min`/`year_max` -- so there is no double-counting. (Checkpoint +#' review docs/phase_r_harmonization_review.md § 0.2.) +#' @noRd +.run_recipe <- function(con, recipe_id, govid, years) { + sql <- sprintf( + "SELECT l.year, l.canonical_govid, + COALESCE(x.gov_name, l.gov_name) AS gov_name, + SUM(l.amt * r.weight) * 1000.0 AS amt_nominal, + string_agg(DISTINCT l.item_code, ',' ORDER BY l.item_code) AS codes_included + FROM long l + JOIN harmonization_recipes r + ON l.item_code = r.component_code + AND l.year BETWEEN r.year_min AND r.year_max + AND (r.gov_type_scope = 'all' + OR (r.gov_type_scope = 'state' AND l.type = 0) + OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3)) + LEFT JOIN canonical_fips_xwalk x USING (canonical_govid) + WHERE r.recipe_id = %1$s + AND l.canonical_govid IN (%2$s) + AND l.year IN (%3$s) + GROUP BY 1, 2, 3 + ORDER BY 1, 2", + .sql_lit_chr(recipe_id), .sql_lit_chr(govid), + paste(as.integer(years), collapse = ",") + ) + result <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) + attr(result, "sql_query") <- sql + result +} + +#' Shape a raw .run_recipe() result into the standard cog_spending()/ +#' cog_revenue() column layout: subtype = "recipe", category = the recipe's +#' label, aggregate_fallback = FALSE (recipes resolve coverage gaps by +#' construction, not by falling back to an aggregate row). +#' @noRd +.shape_recipe_result <- function(result, subtype_col, label) { + sql_query <- attr(result, "sql_query") + n <- nrow(result) + result[[subtype_col]] <- rep("recipe", n) + result$category <- rep(label, n) + result$aggregate_fallback <- rep(FALSE, n) + result <- result[, c( + "year", "canonical_govid", "gov_name", subtype_col, "category", + "amt_nominal", "codes_included", "aggregate_fallback" + ), drop = FALSE] + attr(result, "sql_query") <- sql_query + result +} + +#' Turn a small data.frame into a list-of-lists (one list per row), the +#' shape used for the `recipe$components` provenance block. +#' @noRd +.df_to_row_list <- function(df) { + lapply(seq_len(nrow(df)), function(i) as.list(df[i, , drop = FALSE])) +} diff --git a/R/revenue.R b/R/revenue.R index 5e94d9a..ed71d08 100644 --- a/R/revenue.R +++ b/R/revenue.R @@ -15,7 +15,7 @@ #' @export cog_revenue <- function(govid, years, category = NULL, per_capita = FALSE, adjust_to_year = NULL, - basis = c("harmonized", "raw")) { + basis = c("harmonized", "raw"), recipe = NULL) { .verb_spendrev( verb = "cog_revenue", view_base = "revenue_annotated", @@ -27,6 +27,7 @@ cog_revenue <- function(govid, years, category = NULL, category = category, per_capita = per_capita, adjust_to_year = adjust_to_year, - basis = basis + basis = basis, + recipe = recipe ) } diff --git a/R/series_breaks.R b/R/series_breaks.R new file mode 100644 index 0000000..d4f0c1c --- /dev/null +++ b/R/series_breaks.R @@ -0,0 +1,22 @@ +# R/series_breaks.R +# Populates prov$series_break_refs (schema in inst/schemas/provenance-v1.json +# defines the field; it was always present but always empty pre-Phase-R2) +# with the ids of any catalogued series break whose fin_code appears among +# the result's observed item codes and whose break_year falls inside the +# requested year span -- the "break warnings in the provenance envelope" +# spec § 5 promises downstream consumers (cog-api passes provenance through +# verbatim). schema_version >= 5 only: series_breaks_pq isn't registered on +# an older corpus. + +#' @noRd +.build_series_break_refs <- function(con, codes_observed, years, schema_version) { + if (schema_version < 5L || length(codes_observed) == 0L) return(character(0)) + sql <- sprintf( + "SELECT DISTINCT break_id + FROM series_breaks_pq + WHERE fin_code IN (%s) AND break_year BETWEEN %d AND %d + ORDER BY break_id", + .sql_lit_chr(codes_observed), min(as.integer(years)), max(as.integer(years)) + ) + DBI::dbGetQuery(con, sql)$break_id +} diff --git a/R/spending.R b/R/spending.R index 8526103..70582b6 100644 --- a/R/spending.R +++ b/R/spending.R @@ -29,6 +29,12 @@ #' tables), `basis` silently resolves to `"raw"` when left at its default #' and the resolution is recorded in the provenance; explicitly passing #' `basis = "harmonized"` on such a corpus aborts. +#' @param recipe Optional harmonization recipe id (see [cog_recipes()]) for +#' multi-code cross-vintage series that a 1:1 harmonized_code mapping +#' can't express (e.g. a wide-era aggregate that only splits into leaf +#' codes in the modern era). Mutually exclusive with `category`. The +#' result's subtype column reads `"recipe"` and `category` reads the +#' recipe's label. Requires `schema_version >= 5`. #' @return Tibble with columns `year`, `canonical_govid`, `gov_name`, #' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real`, #' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`, @@ -37,7 +43,7 @@ #' @export cog_spending <- function(govid, years, category = NULL, per_capita = FALSE, adjust_to_year = NULL, - basis = c("harmonized", "raw")) { + basis = c("harmonized", "raw"), recipe = NULL) { .verb_spendrev( verb = "cog_spending", view_base = "spending_annotated", @@ -49,7 +55,8 @@ cog_spending <- function(govid, years, category = NULL, category = category, per_capita = per_capita, adjust_to_year = adjust_to_year, - basis = basis + basis = basis, + recipe = recipe ) } @@ -57,12 +64,13 @@ cog_spending <- function(govid, years, category = NULL, .verb_spendrev <- function(verb, view_base, subtype_col, flow_prefixes, call, govid, years, category, per_capita, adjust_to_year, - basis = c("harmonized", "raw")) { + basis = c("harmonized", "raw"), recipe = NULL) { basis_explicit <- length(basis) == 1L basis <- match.arg(basis, c("harmonized", "raw")) govid <- .coerce_govid_input(govid, arg = "govid") - .validate_verb_inputs(govid, years, category, per_capita, adjust_to_year) + .validate_verb_inputs(govid, years, category, per_capita, adjust_to_year, + recipe) years <- as.integer(years) if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year) @@ -73,9 +81,26 @@ cog_spending <- function(govid, years, category = NULL, resolved <- .resolve_basis(basis, basis_explicit, manifest) - view <- .select_view(view_base, resolved$basis) - sql <- .build_verb_sql(view, subtype_col, govid, years, category) - result <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) + recipe_block <- NULL + category_for_prov <- category + if (!is.null(recipe)) { + .require_schema_v5(con, manifest, "recipe =") + .validate_recipe_id(con, recipe) + comps <- .recipe_components(con, recipe) + recipe_label <- comps$label[[1]] + result <- .run_recipe(con, recipe, govid, years) + sql <- attr(result, "sql_query") + result <- .shape_recipe_result(result, subtype_col, recipe_label) + recipe_block <- list( + recipe_id = recipe, label = recipe_label, + components = .df_to_row_list(comps) + ) + category_for_prov <- recipe_label + } else { + view <- .select_view(view_base, resolved$basis) + sql <- .build_verb_sql(view, subtype_col, govid, years, category) + result <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) + } if (per_capita) result <- .attach_per_capita(result, con, govid) if (!is.null(adjust_to_year)) { @@ -88,12 +113,18 @@ cog_spending <- function(govid, years, category = NULL, con, govid, years, resolved, flow_prefixes ) + suggestions <- if (is.null(recipe)) { + .build_suggestions(con, govid, years, category, result, resolved$basis) + } else { + list() + } + prov <- .build_provenance( verb = verb, call = call, govid = govid, years = years, - category = category, + category = category_for_prov, per_capita = per_capita, adjust_to_year = adjust_to_year, result = result, @@ -101,18 +132,23 @@ cog_spending <- function(govid, years, category = NULL, subtype_col = subtype_col, basis = resolved$basis, basis_note = resolved$note, - harmonization = harmonization + harmonization = harmonization, + recipe = recipe_block, + suggestions = suggestions ) prov$scope$govids_found <- scope$found prov$scope$govids_missing <- scope$missing attr(result, "provenance") <- prov attr(result, ".popyear_range") <- NULL + + if (length(suggestions) > 0L) .inform_suggestions(suggestions) + result } #' @noRd .validate_verb_inputs <- function(govid, years, category, - per_capita, adjust_to_year) { + per_capita, adjust_to_year, recipe = NULL) { if (!is.character(govid) || length(govid) == 0L) { cli::cli_abort("`govid` must be a non-empty character vector.") } @@ -131,6 +167,17 @@ cog_spending <- function(govid, years, category = NULL, cli::cli_abort("`adjust_to_year` must be NULL or a length-1 integer.") } } + if (!is.null(recipe)) { + if (!is.character(recipe) || length(recipe) != 1L) { + cli::cli_abort("`recipe` must be NULL or a length-1 character string.") + } + if (!is.null(category)) { + cli::cli_abort(c( + "`recipe` and `category` are mutually exclusive.", + i = "Pass one or the other, not both." + ), class = "uscogdata_recipe_category_conflict") + } + } invisible(TRUE) } diff --git a/R/suggestions.R b/R/suggestions.R new file mode 100644 index 0000000..fc59d21 --- /dev/null +++ b/R/suggestions.R @@ -0,0 +1,119 @@ +# R/suggestions.R +# Recipe-component-driven signposting: when a basis = "harmonized" query for +# a category comes back with a coverage gap in some requested years (the +# result has no rows at all in that year) that a harmonization recipe would +# actually fill for this government, surface that recipe as a suggestion. +# +# This is deliberately keyed off the recipe catalog's component codes, not +# off harmonization_map rows: no live map row carries a non-blank +# suggested_recipe_id (the corpus's wide era exposes split families like +# corrections functions 04+05 ONLY as aggregate rows, which basis = +# "harmonized" excludes by construction -- there's no NA ruling to hang a +# suggestion off of, just a leaf-code absence a recipe happens to fill). +# See docs/phase_r_harmonization_review.md § 0.3. +# +# Scope is deliberately narrow: signposting only runs when the caller +# supplied a `category` (an un-scoped, all-categories query has no single +# coverage question to answer) and only flags a recipe when the ACTUAL +# result has zero rows in a requested year AND the candidate recipe's own +# generic join (same join .run_recipe() uses, including its wide-era +# aggregate rows) produces at least one row for this government in that +# year. Checking presence per-government (not corpus-wide) avoids false +# positives from ordinary reporting variance -- most governments don't use +# every sibling code in a multi-code category every year, and that is not +# a format-boundary gap worth signposting. + +#' Build the `prov$suggestions` list for a (non-recipe) basis = "harmonized" +#' verb call: recipes whose generic join would fill a real gap in `result`. +#' +#' @param con Active DuckDB connection. +#' @param govid Character vector of canonical_govid values (the verb's raw +#' `govid`). +#' @param years Integer vector of requested years. +#' @param category `category` argument as passed to the verb (character +#' vector or `NULL`; suggestions are only computed when non-NULL). +#' @param result The verb's already-computed result tibble (post basis +#' query, pre per_capita/adjust_to_year). +#' @param basis The *resolved* basis (`"harmonized"` or `"raw"`). +#' @return List of `list(recipe_id, label, available_years, hint)`, possibly +#' empty. +#' @noRd +.build_suggestions <- function(con, govid, years, category, result, basis) { + if (!identical(basis, "harmonized") || is.null(category)) return(list()) + + candidates <- DBI::dbGetQuery(con, sprintf( + "SELECT DISTINCT recipe_id FROM harmonization_recipes + WHERE component_code IN ( + SELECT DISTINCT item_code FROM summary_categories WHERE category IN (%s) + )", + .sql_lit_chr(category) + ))$recipe_id + if (length(candidates) == 0L) return(list()) + + result_years <- if (is.null(result) || nrow(result) == 0L) { + integer(0) + } else { + unique(as.integer(result$year)) + } + gap_years <- setdiff(as.integer(years), result_years) + if (length(gap_years) == 0L) return(list()) + + meta <- tibble::as_tibble(DBI::dbGetQuery(con, sprintf( + "SELECT recipe_id, any_value(label) AS label, + MIN(year_min) AS year_min, MAX(year_max) AS year_max + FROM harmonization_recipes + WHERE recipe_id IN (%s) + GROUP BY recipe_id", + .sql_lit_chr(candidates) + ))) + + # Which (recipe_id, year) pairs the recipe's own generic join actually + # covers for this government, restricted to the gap years -- the same + # join .run_recipe() uses (component year_min/year_max + gov_type_scope, + # no is_aggregate filter), just checking existence instead of summing. + covered <- DBI::dbGetQuery(con, sprintf( + "SELECT DISTINCT r.recipe_id, l.year + FROM long l + JOIN harmonization_recipes r + ON l.item_code = r.component_code + AND l.year BETWEEN r.year_min AND r.year_max + AND (r.gov_type_scope = 'all' + OR (r.gov_type_scope = 'state' AND l.type = 0) + OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3)) + WHERE r.recipe_id IN (%s) + AND l.canonical_govid IN (%s) + AND l.year IN (%s)", + .sql_lit_chr(candidates), .sql_lit_chr(govid), + paste(gap_years, collapse = ",") + )) + + suggestions <- list() + for (rid in candidates) { + if (!rid %in% covered$recipe_id) next + m <- meta[meta$recipe_id == rid, ] + suggestions[[length(suggestions) + 1L]] <- list( + recipe_id = rid, + label = m$label[[1]], + available_years = c(as.integer(m$year_min), as.integer(m$year_max)), + hint = sprintf("re-run with recipe = '%s'", rid) + ) + } + suggestions +} + +#' Emit the single cli::cli_inform() message summarizing all suggestions +#' for a verb call (the brief's "one message", not one per suggestion). +#' Bullet text is pre-formatted plain text (no cli/glue `{}` markup) since +#' recipe ids/labels are untrusted-ish data values, not literal call-site +#' expressions. +#' @noRd +.inform_suggestions <- function(suggestions) { + bullets <- vapply(suggestions, function(s) { + sprintf("%s (%d-%d): %s", s$recipe_id, + s$available_years[1], s$available_years[2], s$hint) + }, character(1)) + cli::cli_inform(c( + i = "Coverage gap detected for the requested years; a harmonization recipe may fill it:", + stats::setNames(bullets, rep("*", length(bullets))) + )) +} diff --git a/man/cog_recipes.Rd b/man/cog_recipes.Rd new file mode 100644 index 0000000..149fb76 --- /dev/null +++ b/man/cog_recipes.Rd @@ -0,0 +1,22 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/recipes.R +\name{cog_recipes} +\alias{cog_recipes} +\title{List available harmonization recipes} +\usage{ +cog_recipes(pattern = NULL) +} +\arguments{ +\item{pattern}{Optional regex matched case-insensitively against +`recipe_id` or `label`.} +} +\value{ +Tibble with columns `recipe_id`, `label`, `n_components`, + `year_min`, `year_max` (the min/max component year coverage), sorted by + `recipe_id`. +} +\description{ +Recipes are multi-code cross-vintage series (see [cog_spending()]'s +`recipe` argument) catalogued in the corpus's `harmonization_recipes` +table. Use this to discover valid `recipe` ids. +} diff --git a/man/cog_revenue.Rd b/man/cog_revenue.Rd index dd1a5e9..18b77df 100644 --- a/man/cog_revenue.Rd +++ b/man/cog_revenue.Rd @@ -10,7 +10,8 @@ cog_revenue( category = NULL, per_capita = FALSE, adjust_to_year = NULL, - basis = c("harmonized", "raw") + basis = c("harmonized", "raw"), + recipe = NULL ) } \arguments{ @@ -40,6 +41,13 @@ folding). On a corpus with `schema_version < 5` (no harmonization tables), `basis` silently resolves to `"raw"` when left at its default and the resolution is recorded in the provenance; explicitly passing `basis = "harmonized"` on such a corpus aborts.} + +\item{recipe}{Optional harmonization recipe id (see [cog_recipes()]) for +multi-code cross-vintage series that a 1:1 harmonized_code mapping +can't express (e.g. a wide-era aggregate that only splits into leaf +codes in the modern era). Mutually exclusive with `category`. The +result's subtype column reads `"recipe"` and `category` reads the +recipe's label. Requires `schema_version >= 5`.} } \value{ Tibble with columns `year`, `canonical_govid`, `gov_name`, diff --git a/man/cog_spending.Rd b/man/cog_spending.Rd index 876a48f..8e4c52a 100644 --- a/man/cog_spending.Rd +++ b/man/cog_spending.Rd @@ -10,7 +10,8 @@ cog_spending( category = NULL, per_capita = FALSE, adjust_to_year = NULL, - basis = c("harmonized", "raw") + basis = c("harmonized", "raw"), + recipe = NULL ) } \arguments{ @@ -40,6 +41,13 @@ folding). On a corpus with `schema_version < 5` (no harmonization tables), `basis` silently resolves to `"raw"` when left at its default and the resolution is recorded in the provenance; explicitly passing `basis = "harmonized"` on such a corpus aborts.} + +\item{recipe}{Optional harmonization recipe id (see [cog_recipes()]) for +multi-code cross-vintage series that a 1:1 harmonized_code mapping +can't express (e.g. a wide-era aggregate that only splits into leaf +codes in the modern era). Mutually exclusive with `category`. The +result's subtype column reads `"recipe"` and `category` reads the +recipe's label. Requires `schema_version >= 5`.} } \value{ Tibble with columns `year`, `canonical_govid`, `gov_name`, diff --git a/tests/testthat/test-explain.R b/tests/testthat/test-explain.R index 83f18ab..e824807 100644 --- a/tests/testthat/test-explain.R +++ b/tests/testthat/test-explain.R @@ -31,6 +31,45 @@ test_that("cog_explain errors on non-verb input", { expect_error(cog_explain(df), "provenance") }) +test_that("cog_explain prints basis + harmonization block", { + skip_if_no_corpus() + r <- cog_spending("121011212191", 2020L, "Corrections") + txt <- paste(c( + capture.output(cog_explain(r)), + capture.output(cog_explain(r), type = "message") + ), collapse = "\n") + expect_true(grepl("Basis: harmonized", txt)) + expect_true(grepl("Harmonization", txt)) + expect_true(grepl("Excluded 0 row", txt)) +}) + +test_that("cog_explain prints a Recipe section for recipe = results", { + skip_if_no_corpus() + r <- cog_spending("121011212191", c(2011L, 2012L), recipe = "corrections_combined") + txt <- paste(c( + capture.output(cog_explain(r)), + capture.output(cog_explain(r), type = "message") + ), collapse = "\n") + expect_true(grepl("Recipe", txt)) + expect_true(grepl("corrections_combined", txt)) + expect_true(grepl("E04", txt)) + expect_true(grepl("E05", txt)) +}) + +test_that("cog_explain prints a Suggestions section when the provenance has one", { + skip_if_no_corpus() + r <- suppressMessages( + cog_spending("121011212191", c(2011L, 2012L), category = "Corrections") + ) + txt <- paste(c( + capture.output(cog_explain(r)), + capture.output(cog_explain(r), type = "message") + ), collapse = "\n") + expect_true(grepl("Suggestions", txt)) + expect_true(grepl("corrections_combined", txt)) + expect_true(grepl("re-run with recipe", txt)) +}) + test_that("cog_explain prints denominator + popyear_range + counts", { skip_if_no_corpus() with_fixture_corpus({ diff --git a/tests/testthat/test-recipes.R b/tests/testthat/test-recipes.R new file mode 100644 index 0000000..44a1bf3 --- /dev/null +++ b/tests/testthat/test-recipes.R @@ -0,0 +1,188 @@ +# tests/testthat/test-recipes.R +# +# cog_recipes(), recipe = in cog_spending()/cog_revenue(), and the +# recipe-component-driven signposting in prov$suggestions (Phase R2 / +# Task 11, schema_version 5). + +test_that("cog_recipes lists the curated catalog including corrections_combined", { + skip_if_no_corpus() + r <- cog_recipes() + expect_s3_class(r, "tbl_df") + expect_equal(names(r), c("recipe_id", "label", "n_components", "year_min", "year_max")) + expect_equal(nrow(r), 24L) + expect_true("corrections_combined" %in% r$recipe_id) + expect_true("t19_selective_sales_wide" %in% r$recipe_id) + expect_true("ig_federal_b89_wide" %in% r$recipe_id) + expect_true("rents_royalties_u4_wide" %in% r$recipe_id) + expect_true("higher_ed_e18_wide" %in% r$recipe_id) + expect_true("cash_securities_z77_wide" %in% r$recipe_id) + # Superseded id from the pre-curation brief text must NOT be present. + expect_false("corrections_judicial_combined" %in% r$recipe_id) +}) + +test_that("cog_recipes(pattern=) filters by recipe_id or label", { + skip_if_no_corpus() + r <- cog_recipes("corrections") + expect_true(nrow(r) >= 1L) + expect_true(all(grepl("corrections", r$recipe_id, ignore.case = TRUE) | + grepl("corrections", r$label, ignore.case = TRUE))) +}) + +test_that("cog_recipes requires schema_version >= 5", { + skip_if_no_corpus() + with_doctored_schema_version(4L, { + expect_error(cog_recipes(), class = "uscogdata_schema_unsupported") + }) +}) + +# --- recipe = : generic join, no is_aggregate filter ----------------------- + +test_that("recipe = 'corrections_combined' is continuous across the 2011->2012 seam", { + skip_if_no_corpus() + r <- cog_spending("121011212191", years = c(2011L, 2012L), + recipe = "corrections_combined") + expect_equal(nrow(r), 2L) + expect_true(all(c("year", "canonical_govid", "gov_name", "spend_subtype", + "category", "amt_nominal", "codes_included", + "aggregate_fallback", "notes") %in% names(r))) + expect_equal(unique(r$spend_subtype), "recipe") + expect_equal(unique(r$category), "Corrections (functions 04+05 combined)") + expect_false(any(r$aggregate_fallback)) + + r2011 <- r$amt_nominal[r$year == 2011L] + r2012 <- r$amt_nominal[r$year == 2012L] + # 2011: E05 only exists as a wide-era AGGREGATE row (is_aggregate = TRUE) + # for Broward -- data-verified $216,088,000. Since .run_recipe() does NOT + # filter is_aggregate (amendment: the recipe join must not, because these + # families exist ONLY as aggregate rows in the wide era), the recipe + # correctly picks this up. + expect_equal(r2011, 216088000) + # 2012: modern E04 leaf ($213,056,000); Broward reports no E05 leaf that + # year, so the recipe total equals E04 alone -- still continuous with the + # 2011 aggregate, proving the wide-aggregate -> modern-leaf handoff. + expect_equal(r2012, 213056000) + expect_true(all(grepl("E04|E05", r$codes_included))) +}) + +test_that("recipe result carries a recipe provenance block with component rows", { + skip_if_no_corpus() + r <- cog_spending("121011212191", years = c(2011L, 2012L), + recipe = "corrections_combined") + prov <- attr(r, "provenance") + expect_equal(prov$basis, "harmonized") + expect_equal(prov$category, "Corrections (functions 04+05 combined)") + expect_type(prov$recipe, "list") + expect_equal(prov$recipe$recipe_id, "corrections_combined") + expect_equal(prov$recipe$label, "Corrections (functions 04+05 combined)") + expect_length(prov$recipe$components, 2L) + comp_codes <- vapply(prov$recipe$components, function(x) x$component_code, character(1)) + expect_setequal(comp_codes, c("E04", "E05")) + # A recipe query resolves its own coverage; it should never also carry + # suggestions for itself. + expect_length(prov$suggestions, 0L) +}) + +test_that("recipe = 't19_selective_sales_wide' sums the local T11/T14 legs when present", { + skip_if_no_corpus() + # Westminster City, CA (canonical_govid 082001211654): T11 = 0 in 2011, + # T11 = 568 (T14 = 0/absent) in 2012 -- a real, data-verified equality/ + # inequality pair inside the amended fixture window (2011-2012), standing + # in for the brief's original 2004/2005 example (out of scope per the + # amended fixture years; the underlying local-tax-split boundary is + # nationally FY2005, but this government's own T11 reporting activates + # within our 2011-2012 window). + r <- cog_revenue("082001211654", years = c(2011L, 2012L), + recipe = "t19_selective_sales_wide") + + # Raw, single-code T19 total (not the "Other Taxes" category total, which + # would also sum in T11/T14/T21/T23/T27/T29/T53/T99 -- queried directly to + # isolate exactly the code the brief's equality/inequality check is about). + con <- uscogdata:::.ensure_session() + raw_t19 <- DBI::dbGetQuery(con, " + SELECT year, SUM(amt) * 1000.0 AS amt + FROM revenue_long + WHERE canonical_govid = '082001211654' AND item_code = 'T19' + AND year IN (2011, 2012) + GROUP BY year ORDER BY year + ") + raw_t19_2011 <- raw_t19$amt[raw_t19$year == 2011L] + raw_t19_2012 <- raw_t19$amt[raw_t19$year == 2012L] + expect_equal(raw_t19_2011, 2231000) + expect_equal(raw_t19_2012, 2365000) + + recipe_2011 <- r$amt_nominal[r$year == 2011L] + recipe_2012 <- r$amt_nominal[r$year == 2012L] + + expect_equal(recipe_2011, raw_t19_2011) # equality: no local T11/T14 yet + expect_gt(recipe_2012, raw_t19_2012) # inequality: local T11 joins in + expect_equal(recipe_2012, raw_t19_2012 + 568000) +}) + +test_that("recipe = and category = together aborts", { + skip_if_no_corpus() + expect_error( + cog_spending("121011212191", 2020L, category = "Corrections", + recipe = "corrections_combined"), + class = "uscogdata_recipe_category_conflict" + ) +}) + +test_that("unknown recipe id aborts and lists valid ids", { + skip_if_no_corpus() + err <- tryCatch( + cog_spending("121011212191", 2020L, recipe = "does_not_exist"), + error = identity + ) + expect_s3_class(err, "uscogdata_unknown_recipe") + expect_match(conditionMessage(err), "corrections_combined") +}) + +test_that("recipe = requires schema_version >= 5", { + skip_if_no_corpus() + with_doctored_schema_version(4L, { + expect_error( + cog_spending("121011212191", 2020L, recipe = "corrections_combined"), + class = "uscogdata_schema_unsupported" + ) + }) +}) + +# --- signposting ------------------------------------------------------- + +test_that("signposting suggests corrections_combined across the 2011->2012 gap", { + skip_if_no_corpus() + expect_message( + r <- cog_spending("121011212191", years = c(2011L, 2012L), + category = "Corrections"), + "recipe" + ) + prov <- attr(r, "provenance") + expect_true(length(prov$suggestions) >= 1L) + ids <- vapply(prov$suggestions, function(s) s$recipe_id, character(1)) + expect_true("corrections_combined" %in% ids) + hit <- prov$suggestions[[which(ids == "corrections_combined")]] + expect_equal(hit$hint, "re-run with recipe = 'corrections_combined'") + expect_equal(hit$available_years, c(1967L, 2023L)) +}) + +test_that("no signposting when the result already has full year coverage", { + skip_if_no_corpus() + r <- cog_spending("121011212191", years = 2019:2020, category = "Corrections") + prov <- attr(r, "provenance") + expect_length(prov$suggestions, 0L) +}) + +test_that("no signposting when category is NULL (unscoped query)", { + skip_if_no_corpus() + r <- cog_spending("121011212191", years = c(2011L, 2012L)) + prov <- attr(r, "provenance") + expect_length(prov$suggestions, 0L) +}) + +test_that("no signposting under basis = 'raw'", { + skip_if_no_corpus() + r <- cog_spending("121011212191", years = c(2011L, 2012L), + category = "Corrections", basis = "raw") + prov <- attr(r, "provenance") + expect_length(prov$suggestions, 0L) +}) diff --git a/tests/testthat/test-spending.R b/tests/testthat/test-spending.R index d6a6c8b..b8e7c08 100644 --- a/tests/testthat/test-spending.R +++ b/tests/testthat/test-spending.R @@ -292,6 +292,21 @@ test_that("v4 corpus: explicit basis = 'harmonized' aborts", { }) }) +test_that("provenance$series_break_refs is a populated-when-applicable character vector", { + skip_if_no_corpus() + with_fixture_corpus({ + r <- cog_spending("121011212191", 2020L, "Corrections") + refs <- attr(r, "provenance")$series_break_refs + expect_type(refs, "character") + # No catalogued series_breaks_pq row falls inside this fixture's + # 2011/2012/2019/2020 window for the codes this query touches (E04/G04) + # -- data-verified; the mechanism itself is what's under test here, via + # a query-shaped unit test in test-views.R since the fixture has no + # positive case to pin against. + expect_equal(refs, character(0)) + }) +}) + test_that("v4 corpus: explicit basis = 'raw' still works", { skip_if_no_corpus() with_doctored_schema_version(4L, { diff --git a/tests/testthat/test-views.R b/tests/testthat/test-views.R index 8cdd86d..fe30aad 100644 --- a/tests/testthat/test-views.R +++ b/tests/testthat/test-views.R @@ -51,6 +51,29 @@ test_that("the REPLACE(harmonized_code AS item_code) pattern folds a collapsed c DBI::dbExecute(con, "DROP TABLE synthetic_long") }) +test_that(".build_series_break_refs matches fin_code + break_year window", { + # No series_breaks_pq row falls inside the bundled fixture's 2011-2020 + # window (data-verified; see the "series_break_refs" test in + # test-spending.R), so this proves the matching logic itself against the + # live view + a synthetic year window that DOES hit a cataloged break + # (SB075, fin_code E62, break_year 2005). + skip_if_no_corpus() + con <- cog_open() + on.exit(cog_close()) + refs <- uscogdata:::.build_series_break_refs( + con, codes_observed = c("E62", "E04"), years = c(2003L, 2006L), + schema_version = 5L + ) + expect_true("SB075" %in% refs) + expect_true("SB071" %in% refs) + + # Gated on schema_version >= 5 even when the codes/years would otherwise match. + refs_v4 <- uscogdata:::.build_series_break_refs( + con, codes_observed = c("E62"), years = c(2003L, 2006L), schema_version = 4L + ) + expect_equal(refs_v4, character(0)) +}) + test_that("schema v5 harmonization views register when the corpus supports them", { skip_if_no_corpus() con <- cog_open()