# R/suggestions.R # Recipe-component-driven signposting: when a basis = "harmonized" query for # a category asks for a code that is itself a harmonization recipe # component, and that specific code has no rows in some requested years # while the recipe's own generic join would still fill those years for this # government, surface that recipe as a suggestion. # # This is deliberately keyed off the recipe catalog's component codes, not # off harmonization_map rows: no live map row carries a non-blank # suggested_recipe_id (the corpus's wide era exposes split families like # corrections functions 04+05 ONLY as aggregate rows, which basis = # "harmonized" excludes by construction -- there's no NA ruling to hang a # suggestion off of, just a leaf-code absence a recipe happens to fill). # See docs/phase_r_harmonization_review.md § 0.3. # # Scope is deliberately narrow in one respect and, as of Phase R3 Task 19c, # deliberately WIDE in another: signposting only runs when the caller # supplied a `category` (an un-scoped, all-categories query has no single # coverage question to answer), but within that category it now checks # EACH recipe component that is itself a category member individually, # rather than asking whether the whole category *result* has zero rows # that year. A recipe fires when one of its own components has zero rows # for this government in a requested year, AND SOME OTHER component of that # SAME recipe -- excluding the gapped one itself -- has a row (same join # .run_recipe() uses, aggregate rows included) for that year. This is the # literal review-doc § 0.3 criterion: "...has no rows ... but other # components do." A component's OWN aggregate-only row does not satisfy # its own gap (self-coverage is not "other components"); only a genuinely # different sibling component can. This fires even if OTHER, unrelated # codes in the same category have full data that year and the overall # result looks complete. That is a deliberate narrowing of the R2-era # false-positive guard: most governments don't use every sibling code in a # multi-code category every year, and per-code detection WILL flag some of # that as a "gap" even though it's really just a government not having # that particular sub-type of spending, not a format-boundary artifact. # The remaining guard against ordinary reporting variance is the # per-government, per-OTHER-component `covered` check below (a component # is only flagged when a DIFFERENT component of the SAME recipe -- not # some unrelated code, and not the gapped component's own aggregate row -- # actually has something to offer in that year); it no longer tries to # avoid noise from sibling *codes*, only from a recipe with genuinely # nothing else to contribute. The acceptable noise level this trade # produces is a product decision, measured (not tuned here) by # data-raw/measure_signposting_rate.R and ruled on at Checkpoint R3. #' Recipe components that are classified under the requested category -- #' the codes a category-scoped query actually "requests". A recipe can #' have components outside the category (e.g. general_gov_e89_wide's E85 #' leg has no category assignment); those never trigger on their own, they #' just were never part of what this query asked for. #' @noRd .category_recipe_components <- function(con, category) { DBI::dbGetQuery(con, sprintf( "SELECT DISTINCT r.recipe_id, r.component_code, r.year_min, r.year_max, r.gov_type_scope FROM harmonization_recipes r JOIN summary_categories sc ON sc.item_code = r.component_code AND sc.category IN (%s)", .sql_lit_chr(category) )) } #' Label + overall year coverage for a set of recipe ids (the suggestion's #' `label`/`available_years`). #' @noRd .recipe_meta <- function(con, candidates) { tibble::as_tibble(DBI::dbGetQuery(con, sprintf( "SELECT recipe_id, any_value(label) AS label, MIN(year_min) AS year_min, MAX(year_max) AS year_max FROM harmonization_recipes WHERE recipe_id IN (%s) GROUP BY recipe_id", .sql_lit_chr(candidates) ))) } #' Which (recipe_id, component_code, year) triples have at least one #' NOT-aggregate row for these governments -- i.e. that specific requested #' code itself has data, scoped exactly like .run_recipe()'s join #' (component year_min/year_max + gov_type_scope). NOT-aggregate mirrors #' what basis = "harmonized" itself excludes: an aggregate-only year is a #' gap for that code exactly as it would be in a plain category query. #' @noRd .component_presence <- function(con, candidates, govid, years_lit) { DBI::dbGetQuery(con, sprintf( "SELECT DISTINCT r.recipe_id, r.component_code, l.year FROM long l JOIN harmonization_recipes r ON l.item_code = r.component_code AND l.year BETWEEN r.year_min AND r.year_max AND (r.gov_type_scope = 'all' OR (r.gov_type_scope = 'state' AND l.type = 0) OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3)) WHERE NOT l.is_aggregate AND r.recipe_id IN (%s) AND l.canonical_govid IN (%s) AND l.year IN (%s)", .sql_lit_chr(candidates), .sql_lit_chr(govid), years_lit )) } #' Which (recipe_id, component_code, year) triples have at least one row #' (aggregate rows included) for these governments -- the same scoping #' .run_recipe()'s join uses (component year_min/year_max + gov_type_scope), #' just checking existence instead of summing. Kept at per-component grain #' (not unioned across the whole recipe, unlike the R2/R3-pre-fix version of #' this function) so a gap check can require the covering evidence to come #' from a DIFFERENT component -- review-doc § 0.3's "other components", not #' the gapped component's own aggregate row. This is the per-government #' guard against ordinary reporting variance: a recipe with genuinely #' nothing to offer from any OTHER component (aggregate or leaf) never #' fires. #' @noRd .recipe_coverage <- function(con, candidates, govid, years_lit) { DBI::dbGetQuery(con, sprintf( "SELECT DISTINCT r.recipe_id, r.component_code, l.year FROM long l JOIN harmonization_recipes r ON l.item_code = r.component_code AND l.year BETWEEN r.year_min AND r.year_max AND (r.gov_type_scope = 'all' OR (r.gov_type_scope = 'state' AND l.type = 0) OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3)) WHERE r.recipe_id IN (%s) AND l.canonical_govid IN (%s) AND l.year IN (%s)", .sql_lit_chr(candidates), .sql_lit_chr(govid), years_lit )) } #' TRUE if recipe `rid` has at least one requested component with an #' in-scope requested year that has no data (`present`), in a year some #' OTHER component of the same recipe is otherwise fillable (`covered`, #' excluding the component under test) -- the per-code gap the R2 #' whole-result check couldn't see, covered by another component the way #' review-doc § 0.3 specifies (not by the gapped component's own aggregate #' row -- that is self-coverage, not "other components", and must not #' count). #' @noRd .recipe_component_gapped <- function(rid, requested, present, covered, years) { comps <- requested[requested$recipe_id == rid, , drop = FALSE] for (i in seq_len(nrow(comps))) { this_code <- comps$component_code[i] in_scope <- years[years >= comps$year_min[i] & years <= comps$year_max[i]] if (length(in_scope) == 0L) next has_data <- present$year[ present$recipe_id == rid & present$component_code == this_code ] gap_years <- setdiff(in_scope, has_data) if (length(gap_years) == 0L) next other_covered_years <- covered$year[ covered$recipe_id == rid & covered$component_code != this_code ] if (any(gap_years %in% other_covered_years)) return(TRUE) } FALSE } #' Build the `prov$suggestions` list for a (non-recipe) basis = "harmonized" #' verb call: recipes whose generic join would fill a real per-code gap for #' the requested category. #' #' @param con Active DuckDB connection. #' @param govid Character vector of canonical_govid values (the verb's raw #' `govid`). #' @param years Integer vector of requested years. #' @param category `category` argument as passed to the verb (character #' vector or `NULL`; suggestions are only computed when non-NULL). #' @param basis The *resolved* basis (`"harmonized"` or `"raw"`). #' @return List of `list(recipe_id, label, available_years, hint)`, possibly #' empty. #' @noRd .build_suggestions <- function(con, govid, years, category, basis) { if (!identical(basis, "harmonized") || is.null(category)) return(list()) requested <- .category_recipe_components(con, category) if (nrow(requested) == 0L) return(list()) candidates <- unique(requested$recipe_id) years_int <- as.integer(years) years_lit <- paste(years_int, collapse = ",") meta <- .recipe_meta(con, candidates) present <- .component_presence(con, candidates, govid, years_lit) covered <- .recipe_coverage(con, candidates, govid, years_lit) suggestions <- list() for (rid in candidates) { if (!.recipe_component_gapped(rid, requested, present, covered, years_int)) next m <- meta[meta$recipe_id == rid, ] suggestions[[length(suggestions) + 1L]] <- list( recipe_id = rid, label = m$label[[1]], available_years = c(as.integer(m$year_min), as.integer(m$year_max)), hint = sprintf("re-run with recipe = '%s'", rid) ) } suggestions } #' Emit the single cli::cli_inform() message summarizing all suggestions #' for a verb call (the brief's "one message", not one per suggestion). #' Bullet text is pre-formatted plain text (no cli/glue `{}` markup) since #' recipe ids/labels are untrusted-ish data values, not literal call-site #' expressions. #' @noRd .inform_suggestions <- function(suggestions) { bullets <- vapply(suggestions, function(s) { sprintf("%s (%d-%d): %s", s$recipe_id, s$available_years[1], s$available_years[2], s$hint) }, character(1)) cli::cli_inform(c( i = "Coverage gap detected for the requested years; a harmonization recipe may fill it:", stats::setNames(bullets, rep("*", length(bullets))) )) }