From 68b72faeae4624d0452f3d79117e3d27be51a04d Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Tue, 28 Apr 2026 10:47:56 -0400 Subject: [PATCH 01/12] refactor(search): rename cog_gov_search() first argument to name Pre-rename in preparation for basket mode. All existing callers in this package and cog_explorer/ pass the first argument positionally, so this rename is non-breaking. No deprecation alias added per design spec (no external consumers; package is pre-release v0.1.0). @param roxygen also updated to match the new formal; man/cog_gov_search.Rd regenerated via devtools::document(). --- R/search.R | 12 ++++++------ man/cog_gov_search.Rd | 4 ++-- 2 files changed, 8 insertions(+), 8 deletions(-) diff --git a/R/search.R b/R/search.R index 01557b1..02ac2c2 100644 --- a/R/search.R +++ b/R/search.R @@ -7,7 +7,7 @@ #' name into one or more `canonical_govid` values before calling #' [cog_spending()] / [cog_revenue()] / etc. #' -#' @param pattern Character regex matched case-insensitively against +#' @param name Character regex matched case-insensitively against #' `gov_name`. `NULL` (default) means no name filter. #' @param state Either a 2-letter USPS abbreviation (e.g. `"FL"`), a FIPS #' integer (e.g. `12`), or `NULL`. @@ -18,7 +18,7 @@ #' @return Tibble from `canonical_fips_xwalk` sorted by `population_acs` #' descending (`NULL`s last). #' @export -cog_gov_search <- function(pattern = NULL, state = NULL, type = NULL) { +cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { if (!is.null(type) && .is_excluded_type(type)) { cli::cli_inform(c( i = "v0.1 covers gov_types 0-3 (state/county/city/township) only.", @@ -29,13 +29,13 @@ cog_gov_search <- function(pattern = NULL, state = NULL, type = NULL) { con <- .ensure_session() preds <- character(0) - if (!is.null(pattern)) { - if (!is.character(pattern) || length(pattern) != 1L) { - cli::cli_abort("`pattern` must be a length-1 character string.") + if (!is.null(name)) { + if (!is.character(name) || length(name) != 1L) { + cli::cli_abort("`name` must be a length-1 character string.") } preds <- c(preds, sprintf("regexp_matches(gov_name, %s, 'i')", - .sql_lit_chr(pattern))) + .sql_lit_chr(name))) } if (!is.null(state)) { st_fips <- .coerce_state_to_fips(state) diff --git a/man/cog_gov_search.Rd b/man/cog_gov_search.Rd index 05f24ee..0d2a0e6 100644 --- a/man/cog_gov_search.Rd +++ b/man/cog_gov_search.Rd @@ -4,10 +4,10 @@ \alias{cog_gov_search} \title{Search for governments by name, state, and/or type} \usage{ -cog_gov_search(pattern = NULL, state = NULL, type = NULL) +cog_gov_search(name = NULL, state = NULL, type = NULL) } \arguments{ -\item{pattern}{Character regex matched case-insensitively against +\item{name}{Character regex matched case-insensitively against `gov_name`. `NULL` (default) means no name filter.} \item{state}{Either a 2-letter USPS abbreviation (e.g. `"FL"`), a FIPS From 7fd29c919da5b7287a6d83a29bb30ab938f8b378 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Tue, 28 Apr 2026 11:02:03 -0400 Subject: [PATCH 02/12] feat(search): add basket-mode argument validator Internal .validate_basket_args() handles length validation and recycling of state/type from length 1. Foundation for basket mode. --- R/search.R | 38 +++++++++++++++++++++ tests/testthat/test-search.R | 64 ++++++++++++++++++++++++++++++++++++ 2 files changed, 102 insertions(+) diff --git a/R/search.R b/R/search.R index 02ac2c2..3b1320b 100644 --- a/R/search.R +++ b/R/search.R @@ -120,3 +120,41 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { WV = "54", WI = "55", WY = "56", AS = "60", GU = "66", MP = "69", PR = "72", VI = "78" ) + +# Validate basket-mode inputs. Returns a list with normalized character +# vectors `name`, `state`, `type`, all of length n = length(name). +# `state` and `type` of length 1 are recycled; lengths must be 1 or n +# otherwise. NULL state/type become a vector of NA_character_. +#' @noRd +.validate_basket_args <- function(name, state, type) { + if (!is.character(name)) { + cli::cli_abort("`name` must be a character vector.") + } + n <- length(name) + + state_norm <- if (is.null(state)) { + rep(NA_character_, n) + } else if (length(state) == 1L) { + rep(as.character(state), n) + } else if (length(state) == n) { + as.character(state) + } else { + cli::cli_abort( + "`state` must be length 1 or {n} (length of `name`); got {length(state)}." + ) + } + + type_norm <- if (is.null(type)) { + rep(NA_character_, n) + } else if (length(type) == 1L) { + rep(as.character(type), n) + } else if (length(type) == n) { + as.character(type) + } else { + cli::cli_abort( + "`type` must be length 1 or {n} (length of `name`); got {length(type)}." + ) + } + + list(name = name, state = state_norm, type = type_norm) +} diff --git a/tests/testthat/test-search.R b/tests/testthat/test-search.R index ddc3ffa..ea19a5f 100644 --- a/tests/testthat/test-search.R +++ b/tests/testthat/test-search.R @@ -72,3 +72,67 @@ test_that("cog_gov_search with no filters returns full registry", { r <- cog_gov_search() expect_gt(nrow(r), 1000L) }) + +# ---- basket mode internal helpers ---- + +test_that(".validate_basket_args recycles state from length 1", { + out <- uscogdata:::.validate_basket_args( + name = c("Broward", "San Diego", "Austin"), + state = "FL", + type = NULL + ) + expect_equal(out$name, c("Broward", "San Diego", "Austin")) + expect_equal(out$state, c("FL", "FL", "FL")) + expect_equal(out$type, c(NA_character_, NA_character_, NA_character_)) +}) + +test_that(".validate_basket_args recycles type from length 1", { + out <- uscogdata:::.validate_basket_args( + name = c("San Diego", "Oakland"), + state = "CA", + type = "city" + ) + expect_equal(out$type, c("city", "city")) +}) + +test_that(".validate_basket_args accepts per-row state and type", { + out <- uscogdata:::.validate_basket_args( + name = c("Broward", "San Diego"), + state = c("FL", "CA"), + type = c(NA, "city") + ) + expect_equal(out$state, c("FL", "CA")) + expect_equal(out$type, c(NA_character_, "city")) +}) + +test_that(".validate_basket_args rejects length-mismatched state", { + expect_error( + uscogdata:::.validate_basket_args( + name = c("Broward", "San Diego", "Austin"), + state = c("FL", "CA"), + type = NULL + ), + regexp = "must be length 1 or 3" + ) +}) + +test_that(".validate_basket_args rejects length-mismatched type", { + expect_error( + uscogdata:::.validate_basket_args( + name = c("Broward", "San Diego"), + state = "FL", + type = c("county", "city", "city") + ), + regexp = "must be length 1 or 2" + ) +}) + +test_that(".validate_basket_args allows NULL state and type", { + out <- uscogdata:::.validate_basket_args( + name = c("Broward", "San Diego"), + state = NULL, + type = NULL + ) + expect_equal(out$state, c(NA_character_, NA_character_)) + expect_equal(out$type, c(NA_character_, NA_character_)) +}) From 4dcf1c72b22bfa6045144a931402146023584477 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Tue, 28 Apr 2026 11:05:14 -0400 Subject: [PATCH 03/12] =?UTF-8?q?feat(search):=20add=20per-row=20resolver?= =?UTF-8?q?=20=E2=80=94=20exact=20match=20branch?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit .resolve_basket_row() handles the exact-match case. Substring fallback and disambiguation branches follow in subsequent commits. --- R/search.R | 50 ++++++++++++++++++++++++++++++++++++ tests/testthat/test-search.R | 32 +++++++++++++++++++++++ 2 files changed, 82 insertions(+) diff --git a/R/search.R b/R/search.R index 3b1320b..b2b5127 100644 --- a/R/search.R +++ b/R/search.R @@ -158,3 +158,53 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { list(name = name, state = state_norm, type = type_norm) } + +# Resolve a single basket-mode input row. Returns a list with components: +# status : "resolved" | "largest_pop" | "ambiguous" | "no_match" +# match_method : "exact" | "substring" | NA_character_ +# n_candidates : int +# row : tibble (single resolved row, or 0-row tibble for unresolved) +# candidates : tibble (all rows that matched, for sidecar) +# Internal use only; takes an active DuckDB connection to reuse the session. +#' @noRd +.resolve_basket_row <- function(name, state, type, con) { + preds <- character(0) + if (!is.na(state)) { + st_fips <- .coerce_state_to_fips(state) + preds <- c(preds, sprintf("fips_state = %s", .sql_lit_chr(st_fips))) + } + if (!is.na(type)) { + int_type <- .coerce_type(type) + preds <- c(preds, sprintf("govs_type = %d", int_type)) + } + + base_where <- if (length(preds) == 0L) "" else paste("WHERE", paste(preds, collapse = " AND ")) + + exact_sql <- paste( + "SELECT * FROM canonical_fips_xwalk", + base_where, + if (nzchar(base_where)) "AND" else "WHERE", + sprintf("LOWER(gov_name) = LOWER(%s)", .sql_lit_chr(name)) + ) + exact <- tibble::as_tibble(DBI::dbGetQuery(con, exact_sql)) + + if (nrow(exact) == 1L) { + return(list( + status = "resolved", + match_method = "exact", + n_candidates = 1L, + row = exact, + candidates = exact + )) + } + + # Substring + disambiguation branches added in subsequent tasks. + empty <- exact[0, , drop = FALSE] + list( + status = "no_match", + match_method = NA_character_, + n_candidates = 0L, + row = empty, + candidates = empty + ) +} diff --git a/tests/testthat/test-search.R b/tests/testthat/test-search.R index ea19a5f..b5f68b3 100644 --- a/tests/testthat/test-search.R +++ b/tests/testthat/test-search.R @@ -136,3 +136,35 @@ test_that(".validate_basket_args allows NULL state and type", { expect_equal(out$state, c(NA_character_, NA_character_)) expect_equal(out$type, c(NA_character_, NA_character_)) }) + +test_that(".resolve_basket_row exact match returns one row", { + con <- uscogdata:::.ensure_session() + out <- uscogdata:::.resolve_basket_row( + name = "BROWARD COUNTY", state = "FL", type = NA_character_, con = con + ) + expect_equal(out$status, "resolved") + expect_equal(out$match_method, "exact") + expect_equal(out$n_candidates, 1L) + expect_equal(nrow(out$row), 1L) + expect_equal(out$row$canonical_govid, "101006006") + expect_equal(out$row$gov_name, "BROWARD COUNTY") +}) + +test_that(".resolve_basket_row exact match is case-insensitive", { + con <- uscogdata:::.ensure_session() + out <- uscogdata:::.resolve_basket_row( + name = "broward county", state = "FL", type = NA_character_, con = con + ) + expect_equal(out$status, "resolved") + expect_equal(out$match_method, "exact") + expect_equal(out$row$canonical_govid, "101006006") +}) + +test_that(".resolve_basket_row exact match honors per-row type", { + con <- uscogdata:::.ensure_session() + out <- uscogdata:::.resolve_basket_row( + name = "SAN DIEGO CITY", state = "CA", type = "city", con = con + ) + expect_equal(out$status, "resolved") + expect_equal(out$row$canonical_govid, "052037010") +}) From 392e818f525c1c406c0498ba8613f4380aed8a86 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Tue, 28 Apr 2026 11:06:30 -0400 Subject: [PATCH 04/12] feat(search): add substring fallback and no_match handling .resolve_basket_row() now falls back to case-insensitive substring match when no exact match is found, and short-circuits empty/whitespace input to no_match. Disambiguation stub raises pending Task 5. --- R/search.R | 69 ++++++++++++++++++++++++++++-------- tests/testthat/test-search.R | 36 +++++++++++++++++++ 2 files changed, 90 insertions(+), 15 deletions(-) diff --git a/R/search.R b/R/search.R index b2b5127..67a7672 100644 --- a/R/search.R +++ b/R/search.R @@ -168,6 +168,18 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { # Internal use only; takes an active DuckDB connection to reuse the session. #' @noRd .resolve_basket_row <- function(name, state, type, con) { + # Short-circuit: empty/whitespace name -> no_match without SQL. + if (!nzchar(trimws(name))) { + empty <- .empty_xwalk_tibble() + return(list( + status = "no_match", + match_method = NA_character_, + n_candidates = 0L, + row = empty, + candidates = empty + )) + } + preds <- character(0) if (!is.na(state)) { st_fips <- .coerce_state_to_fips(state) @@ -177,34 +189,61 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { int_type <- .coerce_type(type) preds <- c(preds, sprintf("govs_type = %d", int_type)) } - base_where <- if (length(preds) == 0L) "" else paste("WHERE", paste(preds, collapse = " AND ")) + conj <- if (nzchar(base_where)) "AND" else "WHERE" exact_sql <- paste( "SELECT * FROM canonical_fips_xwalk", base_where, - if (nzchar(base_where)) "AND" else "WHERE", + conj, sprintf("LOWER(gov_name) = LOWER(%s)", .sql_lit_chr(name)) ) exact <- tibble::as_tibble(DBI::dbGetQuery(con, exact_sql)) if (nrow(exact) == 1L) { return(list( - status = "resolved", - match_method = "exact", - n_candidates = 1L, - row = exact, - candidates = exact + status = "resolved", + match_method = "exact", + n_candidates = 1L, + row = exact, + candidates = exact )) } + if (nrow(exact) > 1L) { + return(.disambiguate(exact, method = "exact")) + } - # Substring + disambiguation branches added in subsequent tasks. - empty <- exact[0, , drop = FALSE] - list( - status = "no_match", - match_method = NA_character_, - n_candidates = 0L, - row = empty, - candidates = empty + sub_sql <- paste( + "SELECT * FROM canonical_fips_xwalk", + base_where, + conj, + sprintf("regexp_matches(gov_name, %s, 'i')", .sql_lit_chr(name)) ) + sub <- tibble::as_tibble(DBI::dbGetQuery(con, sub_sql)) + + if (nrow(sub) == 0L) { + return(list( + status = "no_match", + match_method = NA_character_, + n_candidates = 0L, + row = sub, + candidates = sub + )) + } + if (nrow(sub) == 1L) { + return(list( + status = "resolved", + match_method = "substring", + n_candidates = 1L, + row = sub, + candidates = sub + )) + } + .disambiguate(sub, method = "substring") +} + +# Stub for Task 5; raises so any accidental hit during Task 4 is loud. +#' @noRd +.disambiguate <- function(matches, method) { + cli::cli_abort("internal: .disambiguate() not yet implemented") } diff --git a/tests/testthat/test-search.R b/tests/testthat/test-search.R index b5f68b3..5076a7f 100644 --- a/tests/testthat/test-search.R +++ b/tests/testthat/test-search.R @@ -168,3 +168,39 @@ test_that(".resolve_basket_row exact match honors per-row type", { expect_equal(out$status, "resolved") expect_equal(out$row$canonical_govid, "052037010") }) + +test_that(".resolve_basket_row substring fallback resolves single match", { + con <- uscogdata:::.ensure_session() + out <- uscogdata:::.resolve_basket_row( + name = "Broward", state = "FL", type = NA_character_, con = con + ) + expect_equal(out$status, "resolved") + expect_equal(out$match_method, "substring") + expect_equal(out$n_candidates, 1L) + expect_equal(out$row$canonical_govid, "101006006") +}) + +test_that(".resolve_basket_row no_match returns 0-row tibble", { + con <- uscogdata:::.ensure_session() + out <- uscogdata:::.resolve_basket_row( + name = "Notarealplace", state = "NY", type = NA_character_, con = con + ) + expect_equal(out$status, "no_match") + expect_true(is.na(out$match_method)) + expect_equal(out$n_candidates, 0L) + expect_equal(nrow(out$row), 0L) + expect_equal(nrow(out$candidates), 0L) +}) + +test_that(".resolve_basket_row treats empty/whitespace name as no_match", { + con <- uscogdata:::.ensure_session() + out_empty <- uscogdata:::.resolve_basket_row( + name = "", state = "FL", type = NA_character_, con = con + ) + expect_equal(out_empty$status, "no_match") + + out_ws <- uscogdata:::.resolve_basket_row( + name = " ", state = "FL", type = NA_character_, con = con + ) + expect_equal(out_ws$status, "no_match") +}) From 2f47f7ae673a7014118311aea9a379183e572042 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Tue, 28 Apr 2026 11:08:26 -0400 Subject: [PATCH 05/12] =?UTF-8?q?feat(search):=20add=20disambiguation=20?= =?UTF-8?q?=E2=80=94=20largest=5Fpop=20and=20ambiguous=20branches?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Multi-row matches within a single govs_type pick the largest-population row (status=largest_pop). Multi-row matches spanning >=2 types return no basket row (status=ambiguous) with all candidates preserved for the sidecar. --- R/search.R | 23 +++++++++++++++++++-- tests/testthat/test-search.R | 39 ++++++++++++++++++++++++++++++++++++ 2 files changed, 60 insertions(+), 2 deletions(-) diff --git a/R/search.R b/R/search.R index 67a7672..e124c48 100644 --- a/R/search.R +++ b/R/search.R @@ -242,8 +242,27 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { .disambiguate(sub, method = "substring") } -# Stub for Task 5; raises so any accidental hit during Task 4 is loud. +# Disambiguate a multi-row match set. Either picks the largest-pop row +# (within single-type) or returns an ambiguous result with no basket row. #' @noRd .disambiguate <- function(matches, method) { - cli::cli_abort("internal: .disambiguate() not yet implemented") + types <- unique(matches$govs_type) + if (length(types) == 1L) { + pick <- matches[order(-matches$population_acs, na.last = TRUE), , drop = FALSE][1L, , drop = FALSE] + return(list( + status = "largest_pop", + match_method = method, + n_candidates = nrow(matches), + row = pick, + candidates = matches + )) + } + empty <- matches[0, , drop = FALSE] + list( + status = "ambiguous", + match_method = NA_character_, + n_candidates = nrow(matches), + row = empty, + candidates = matches + ) } diff --git a/tests/testthat/test-search.R b/tests/testthat/test-search.R index 5076a7f..2c7b843 100644 --- a/tests/testthat/test-search.R +++ b/tests/testthat/test-search.R @@ -204,3 +204,42 @@ test_that(".resolve_basket_row treats empty/whitespace name as no_match", { ) expect_equal(out_ws$status, "no_match") }) + +test_that(".resolve_basket_row largest_pop within single type", { + # FL Miami substring matches 10 cities (all govs_type = 2), largest pop + # is MIAMI CITY at 443665. + con <- uscogdata:::.ensure_session() + out <- uscogdata:::.resolve_basket_row( + name = "Miami", state = "FL", type = NA_character_, con = con + ) + expect_equal(out$status, "largest_pop") + expect_equal(out$match_method, "substring") + expect_gte(out$n_candidates, 2L) + expect_equal(out$row$canonical_govid, "102013013") + expect_equal(out$row$gov_name, "MIAMI CITY") +}) + +test_that(".resolve_basket_row ambiguous across types", { + # SAN DIEGO substring matches both SAN DIEGO COUNTY (type 1) and + # SAN DIEGO CITY (type 2) in CA. + con <- uscogdata:::.ensure_session() + out <- uscogdata:::.resolve_basket_row( + name = "San Diego", state = "CA", type = NA_character_, con = con + ) + expect_equal(out$status, "ambiguous") + expect_true(is.na(out$match_method)) + expect_equal(out$n_candidates, 2L) + expect_equal(nrow(out$row), 0L) + expect_equal(nrow(out$candidates), 2L) + expect_setequal(out$candidates$govs_type, c(1L, 2L)) +}) + +test_that(".resolve_basket_row resolves with type override on ambiguous case", { + con <- uscogdata:::.ensure_session() + out <- uscogdata:::.resolve_basket_row( + name = "San Diego", state = "CA", type = "city", con = con + ) + expect_equal(out$status, "resolved") + expect_equal(out$match_method, "substring") + expect_equal(out$row$canonical_govid, "052037010") +}) From 98ed6318a62bb76f11023f9b0209105cdd7371ac Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Tue, 28 Apr 2026 11:39:53 -0400 Subject: [PATCH 06/12] feat(search): basket mode for cog_gov_search() Vector name + state + type arguments dispatch to a per-row resolver that produces a basket tibble with a 'resolution' sidecar attribute. Utility mode (length-1 name) is unchanged. --- R/search.R | 110 ++++++++++++++++++++++++++++++----- tests/testthat/test-search.R | 97 ++++++++++++++++++++++++++++++ 2 files changed, 193 insertions(+), 14 deletions(-) diff --git a/R/search.R b/R/search.R index e124c48..dbccc0a 100644 --- a/R/search.R +++ b/R/search.R @@ -2,32 +2,45 @@ #' Search for governments by name, state, and/or type #' -#' Returns rows from `canonical_fips_xwalk` matching the supplied filters. -#' Intended as the entry point users call to resolve a human-readable place -#' name into one or more `canonical_govid` values before calling -#' [cog_spending()] / [cog_revenue()] / etc. +#' Two modes: #' -#' @param name Character regex matched case-insensitively against -#' `gov_name`. `NULL` (default) means no name filter. -#' @param state Either a 2-letter USPS abbreviation (e.g. `"FL"`), a FIPS -#' integer (e.g. `12`), or `NULL`. +#' * **Utility mode** (single `name`): returns all rows from +#' `canonical_fips_xwalk` whose `gov_name` matches the regex +#' case-insensitively, sorted by `population_acs` descending. +#' * **Basket mode** (`length(name) > 1`): resolves each input row to a +#' single canonical govid via exact-then-substring matching with +#' deterministic disambiguation. Returns up to `length(name)` rows in +#' input order plus a `"resolution"` sidecar attribute. See +#' [cog_basket_resolution()]. +#' +#' @param name Character vector of place name(s). Length 1 = utility mode; +#' length >1 = basket mode. +#' @param state Either a 2-letter USPS abbreviation, a FIPS integer, or +#' `NULL`. Length 1 recycles across all entries in basket mode. #' @param type Government type: an integer in `0:3` or one of `"state"`, -#' `"county"`, `"city"`, `"township"`. Passing `4`, `5`, -#' `"special_district"`, or `"school_district"` emits an explanatory -#' message and returns an empty tibble (v0.1 corpus excludes those types). -#' @return Tibble from `canonical_fips_xwalk` sorted by `population_acs` -#' descending (`NULL`s last). +#' `"county"`, `"city"`, `"township"`, or `NA` (per-row optional in +#' basket mode). Passing `4`, `5`, `"special_district"`, or +#' `"school_district"` emits an explanatory message and returns an +#' empty tibble (v0.1 corpus excludes those types). +#' @return Tibble from `canonical_fips_xwalk`. In utility mode, sorted by +#' `population_acs` descending (`NULL`s last). In basket mode, in input +#' order, with `attr(result, "resolution")` set to the sidecar tibble. #' @export cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { - if (!is.null(type) && .is_excluded_type(type)) { + if (!is.null(type) && length(type) == 1L && .is_excluded_type(type)) { cli::cli_inform(c( i = "v0.1 covers gov_types 0-3 (state/county/city/township) only.", i = "Types 4 (special districts) and 5 (school districts) are excluded; see vignette('coverage-scope')." )) return(.empty_xwalk_tibble()) } + con <- .ensure_session() + if (length(name) > 1L) { + return(.resolve_basket(name = name, state = state, type = type, con = con)) + } + preds <- character(0) if (!is.null(name)) { if (!is.character(name) || length(name) != 1L) { @@ -266,3 +279,72 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { candidates = matches ) } + +# Orchestrates basket-mode resolution: validate, per-row resolve, +# assemble the basket tibble + sidecar, attach the sidecar as an attr. +# Caller is responsible for emitting any post-resolution summary message +# (see Task 7 — this stays silent for now). +#' @noRd +.resolve_basket <- function(name, state, type, con) { + args <- .validate_basket_args(name = name, state = state, type = type) + n <- length(args$name) + + resolved <- vector("list", n) + for (i in seq_len(n)) { + resolved[[i]] <- .resolve_basket_row( + name = args$name[i], + state = args$state[i], + type = args$type[i], + con = con + ) + } + + basket_rows <- lapply(resolved, function(r) r$row) + basket <- dplyr::bind_rows(basket_rows[vapply(basket_rows, function(r) nrow(r) > 0L, logical(1))]) + if (nrow(basket) == 0L) basket <- .empty_xwalk_tibble() + + sidecar <- .build_sidecar(args, resolved) + attr(basket, "resolution") <- sidecar + basket +} + +# Build the sidecar tibble. One row per input; carries query_*, status, +# match_method, canonical_govid, gov_name, n_candidates, and a list-col +# `candidates` of full-schema match-candidate tibbles. +#' @noRd +.build_sidecar <- function(args, resolved) { + type_label <- unname(vapply(args$type, function(t) { + if (is.na(t)) NA_character_ else .type_to_label(t) + }, character(1))) + + status <- vapply(resolved, `[[`, character(1), "status") + method <- vapply(resolved, `[[`, character(1), "match_method") + ncand <- vapply(resolved, `[[`, integer(1), "n_candidates") + govid <- vapply(resolved, function(r) { + if (nrow(r$row) == 0L) NA_character_ else r$row$canonical_govid[1L] + }, character(1)) + gname <- vapply(resolved, function(r) { + if (nrow(r$row) == 0L) NA_character_ else r$row$gov_name[1L] + }, character(1)) + cands <- lapply(resolved, `[[`, "candidates") + + tibble::tibble( + query_name = args$name, + query_state = args$state, + query_type = type_label, + status = status, + match_method = method, + canonical_govid = govid, + gov_name = gname, + n_candidates = ncand, + candidates = cands + ) +} + +# Convert a type input (integer-like or label) into the canonical label +# string used in the sidecar query_type column. +#' @noRd +.type_to_label <- function(type) { + int_type <- .coerce_type(type) + c("0" = "state", "1" = "county", "2" = "city", "3" = "township")[[as.character(int_type)]] +} diff --git a/tests/testthat/test-search.R b/tests/testthat/test-search.R index 2c7b843..f8f9ca2 100644 --- a/tests/testthat/test-search.R +++ b/tests/testthat/test-search.R @@ -243,3 +243,100 @@ test_that(".resolve_basket_row resolves with type override on ambiguous case", { expect_equal(out$match_method, "substring") expect_equal(out$row$canonical_govid, "052037010") }) + +# ---- basket mode public surface ---- + +test_that("cog_gov_search basket mode resolves clean inputs in input order", { + skip_if_no_corpus() + basket <- cog_gov_search( + name = c("BROWARD COUNTY", "SAN DIEGO CITY", "AUSTIN CITY"), + state = c("FL", "CA", "TX") + ) + expect_s3_class(basket, "tbl_df") + expect_equal(nrow(basket), 3L) + expect_equal(basket$canonical_govid, c("101006006", "052037010", "442227001")) + expect_equal(basket$gov_name, c("BROWARD COUNTY", "SAN DIEGO CITY", "AUSTIN CITY")) +}) + +test_that("cog_gov_search basket mode attaches a resolution sidecar", { + skip_if_no_corpus() + basket <- cog_gov_search( + name = c("Broward", "San Diego"), + state = c("FL", "CA"), + type = c(NA, "city") + ) + res <- attr(basket, "resolution") + expect_s3_class(res, "tbl_df") + expect_equal(nrow(res), 2L) + expect_equal(res$query_name, c("Broward", "San Diego")) + expect_equal(res$query_state, c("FL", "CA")) + expect_equal(res$query_type, c(NA_character_, "city")) + expect_equal(res$status, c("resolved", "resolved")) + expect_equal(res$match_method, c("substring", "substring")) + expect_true(is.list(res$candidates)) +}) + +test_that("cog_gov_search basket mode skips ambiguous and no_match rows", { + skip_if_no_corpus() + basket <- suppressMessages(cog_gov_search( + name = c("Broward", "San Diego", "Notarealplace"), + state = c("FL", "CA", "NY") + )) + # Broward resolves; San Diego ambiguous; Notarealplace no_match. + expect_equal(nrow(basket), 1L) + expect_equal(basket$canonical_govid, "101006006") + res <- attr(basket, "resolution") + expect_equal(nrow(res), 3L) + expect_equal(res$status, c("resolved", "ambiguous", "no_match")) +}) + +test_that("cog_gov_search basket mode preserves input order", { + skip_if_no_corpus() + basket <- cog_gov_search( + name = c("AUSTIN CITY", "BROWARD COUNTY", "SAN DIEGO CITY"), + state = c("TX", "FL", "CA") + ) + expect_equal(basket$gov_name, c("AUSTIN CITY", "BROWARD COUNTY", "SAN DIEGO CITY")) +}) + +test_that("cog_gov_search basket mode recycles single state", { + skip_if_no_corpus() + basket <- cog_gov_search( + name = c("SAN DIEGO CITY", "OAKLAND CITY"), + state = "CA" + ) + expect_equal(nrow(basket), 2L) + expect_equal(basket$canonical_govid, c("052037010", "052001009")) +}) + +test_that("cog_gov_search basket mode within-type largest_pop records candidates", { + skip_if_no_corpus() + basket <- suppressMessages(cog_gov_search( + name = c("Miami", "OAKLAND CITY"), + state = c("FL", "CA") + )) + expect_equal(nrow(basket), 2L) + res <- attr(basket, "resolution") + miami_row <- res[res$query_name == "Miami", ] + expect_equal(miami_row$status, "largest_pop") + expect_equal(miami_row$canonical_govid, "102013013") + expect_gte(miami_row$n_candidates, 2L) + expect_gte(nrow(miami_row$candidates[[1]]), 2L) +}) + +test_that("cog_gov_search utility mode (length-1 name) has no sidecar", { + skip_if_no_corpus() + r <- cog_gov_search("BROWARD") + expect_null(attr(r, "resolution")) + expect_gte(nrow(r), 1L) +}) + +test_that("cog_gov_search basket mode validates argument lengths", { + expect_error( + cog_gov_search( + name = c("Broward", "San Diego", "Austin"), + state = c("FL", "CA") + ), + regexp = "must be length 1 or 3" + ) +}) From d9bf0552b7d69827b7c4d6ddfd2a882966b428bc Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Tue, 28 Apr 2026 11:41:28 -0400 Subject: [PATCH 07/12] feat(search): post-resolution summary message in basket mode Emits a single cli::cli_inform when any input was ambiguous, missed, or fell back to largest-population. Silent on clean baskets. --- R/search.R | 30 ++++++++++++++++++++++++++++-- tests/testthat/test-search.R | 35 +++++++++++++++++++++++++++++++++++ 2 files changed, 63 insertions(+), 2 deletions(-) diff --git a/R/search.R b/R/search.R index dbccc0a..f9e169c 100644 --- a/R/search.R +++ b/R/search.R @@ -282,8 +282,6 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { # Orchestrates basket-mode resolution: validate, per-row resolve, # assemble the basket tibble + sidecar, attach the sidecar as an attr. -# Caller is responsible for emitting any post-resolution summary message -# (see Task 7 — this stays silent for now). #' @noRd .resolve_basket <- function(name, state, type, con) { args <- .validate_basket_args(name = name, state = state, type = type) @@ -305,6 +303,7 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { sidecar <- .build_sidecar(args, resolved) attr(basket, "resolution") <- sidecar + .basket_summary_message(sidecar) basket } @@ -348,3 +347,30 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { int_type <- .coerce_type(type) c("0" = "state", "1" = "county", "2" = "city", "3" = "township")[[as.character(int_type)]] } + +# Single post-resolution summary message. Silent on clean baskets; +# emits one cli_inform with two-line body otherwise. +#' @noRd +.basket_summary_message <- function(sidecar) { + status <- sidecar$status + n_input <- length(status) + n_basket <- sum(status %in% c("resolved", "largest_pop")) + n_amb <- sum(status == "ambiguous") + n_nm <- sum(status == "no_match") + n_lp <- sum(status == "largest_pop") + + if (n_amb == 0L && n_nm == 0L && n_lp == 0L) return(invisible(NULL)) + + parts <- c( + if (n_amb > 0L) sprintf("%d ambiguous", n_amb), + if (n_nm > 0L) sprintf("%d with no match", n_nm), + if (n_lp > 0L) sprintf("%d used largest-population fallback", n_lp) + ) + + cli::cli_inform(c( + i = sprintf("Basket resolved %d of %d entries.", n_basket, n_input), + i = paste(parts, collapse = ", "), + i = "Inspect with `cog_basket_resolution(result)` or filter to problem rows with `cog_basket_unresolved(result)`." + )) + invisible(NULL) +} diff --git a/tests/testthat/test-search.R b/tests/testthat/test-search.R index f8f9ca2..ef2a892 100644 --- a/tests/testthat/test-search.R +++ b/tests/testthat/test-search.R @@ -340,3 +340,38 @@ test_that("cog_gov_search basket mode validates argument lengths", { regexp = "must be length 1 or 3" ) }) + +# ---- basket mode summary message ---- + +test_that("cog_gov_search basket mode is silent on clean basket", { + skip_if_no_corpus() + expect_message( + cog_gov_search( + name = c("BROWARD COUNTY", "SAN DIEGO CITY"), + state = c("FL", "CA") + ), + regexp = NA # NA = expect no message + ) +}) + +test_that("cog_gov_search basket mode reports breakdown on partial basket", { + skip_if_no_corpus() + expect_message( + cog_gov_search( + name = c("Broward", "San Diego", "Notarealplace"), + state = c("FL", "CA", "NY") + ), + regexp = "Basket resolved 1 of 3" + ) +}) + +test_that("cog_gov_search basket mode message points to the sidecar accessor", { + skip_if_no_corpus() + expect_message( + cog_gov_search( + name = c("Broward", "Notarealplace"), + state = c("FL", "NY") + ), + regexp = "cog_basket_resolution" + ) +}) From ae04d4f48fc6b67e1b580b0515786b0bc0d1cfe5 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Tue, 28 Apr 2026 11:53:40 -0400 Subject: [PATCH 08/12] refactor(search): split sidecar helpers into R/basket.R Moves .build_sidecar(), .type_to_label(), .basket_summary_message() out of R/search.R and into a new R/basket.R. No behavior change; brings R/search.R back under its 350-line budget. R/basket.R will also host the public sidecar accessors added in the next commit. --- R/basket.R | 74 ++++++++++++++++++++++++++++++++++++++++++++++++++++++ R/search.R | 68 ------------------------------------------------- 2 files changed, 74 insertions(+), 68 deletions(-) create mode 100644 R/basket.R diff --git a/R/basket.R b/R/basket.R new file mode 100644 index 0000000..0f9d91a --- /dev/null +++ b/R/basket.R @@ -0,0 +1,74 @@ +# R/basket.R +# +# Internals supporting the basket-mode sidecar (constructed in +# .resolve_basket() — see R/search.R) plus the user-facing accessors +# cog_basket_resolution() and cog_basket_unresolved() (added in a +# later step). + +# Build the sidecar tibble. One row per input; carries query_*, status, +# match_method, canonical_govid, gov_name, n_candidates, and a list-col +# `candidates` of full-schema match-candidate tibbles. +#' @noRd +.build_sidecar <- function(args, resolved) { + type_label <- unname(vapply(args$type, function(t) { + if (is.na(t)) NA_character_ else .type_to_label(t) + }, character(1))) + + status <- vapply(resolved, `[[`, character(1), "status") + method <- vapply(resolved, `[[`, character(1), "match_method") + ncand <- vapply(resolved, `[[`, integer(1), "n_candidates") + govid <- vapply(resolved, function(r) { + if (nrow(r$row) == 0L) NA_character_ else r$row$canonical_govid[1L] + }, character(1)) + gname <- vapply(resolved, function(r) { + if (nrow(r$row) == 0L) NA_character_ else r$row$gov_name[1L] + }, character(1)) + cands <- lapply(resolved, `[[`, "candidates") + + tibble::tibble( + query_name = args$name, + query_state = args$state, + query_type = type_label, + status = status, + match_method = method, + canonical_govid = govid, + gov_name = gname, + n_candidates = ncand, + candidates = cands + ) +} + +# Convert a type input (integer-like or label) into the canonical label +# string used in the sidecar query_type column. +#' @noRd +.type_to_label <- function(type) { + int_type <- .coerce_type(type) + unname(c("0" = "state", "1" = "county", "2" = "city", "3" = "township")[[as.character(int_type)]]) +} + +# Single post-resolution summary message. Silent on clean baskets; +# emits one cli_inform with two-line body otherwise. +#' @noRd +.basket_summary_message <- function(sidecar) { + status <- sidecar$status + n_input <- length(status) + n_basket <- sum(status %in% c("resolved", "largest_pop")) + n_amb <- sum(status == "ambiguous") + n_nm <- sum(status == "no_match") + n_lp <- sum(status == "largest_pop") + + if (n_amb == 0L && n_nm == 0L && n_lp == 0L) return(invisible(NULL)) + + parts <- c( + if (n_amb > 0L) sprintf("%d ambiguous", n_amb), + if (n_nm > 0L) sprintf("%d with no match", n_nm), + if (n_lp > 0L) sprintf("%d used largest-population fallback", n_lp) + ) + + cli::cli_inform(c( + i = sprintf("Basket resolved %d of %d entries.", n_basket, n_input), + i = paste(parts, collapse = ", "), + i = "Inspect with `cog_basket_resolution(result)` or filter to problem rows with `cog_basket_unresolved(result)`." + )) + invisible(NULL) +} diff --git a/R/search.R b/R/search.R index f9e169c..83f0c6c 100644 --- a/R/search.R +++ b/R/search.R @@ -306,71 +306,3 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { .basket_summary_message(sidecar) basket } - -# Build the sidecar tibble. One row per input; carries query_*, status, -# match_method, canonical_govid, gov_name, n_candidates, and a list-col -# `candidates` of full-schema match-candidate tibbles. -#' @noRd -.build_sidecar <- function(args, resolved) { - type_label <- unname(vapply(args$type, function(t) { - if (is.na(t)) NA_character_ else .type_to_label(t) - }, character(1))) - - status <- vapply(resolved, `[[`, character(1), "status") - method <- vapply(resolved, `[[`, character(1), "match_method") - ncand <- vapply(resolved, `[[`, integer(1), "n_candidates") - govid <- vapply(resolved, function(r) { - if (nrow(r$row) == 0L) NA_character_ else r$row$canonical_govid[1L] - }, character(1)) - gname <- vapply(resolved, function(r) { - if (nrow(r$row) == 0L) NA_character_ else r$row$gov_name[1L] - }, character(1)) - cands <- lapply(resolved, `[[`, "candidates") - - tibble::tibble( - query_name = args$name, - query_state = args$state, - query_type = type_label, - status = status, - match_method = method, - canonical_govid = govid, - gov_name = gname, - n_candidates = ncand, - candidates = cands - ) -} - -# Convert a type input (integer-like or label) into the canonical label -# string used in the sidecar query_type column. -#' @noRd -.type_to_label <- function(type) { - int_type <- .coerce_type(type) - c("0" = "state", "1" = "county", "2" = "city", "3" = "township")[[as.character(int_type)]] -} - -# Single post-resolution summary message. Silent on clean baskets; -# emits one cli_inform with two-line body otherwise. -#' @noRd -.basket_summary_message <- function(sidecar) { - status <- sidecar$status - n_input <- length(status) - n_basket <- sum(status %in% c("resolved", "largest_pop")) - n_amb <- sum(status == "ambiguous") - n_nm <- sum(status == "no_match") - n_lp <- sum(status == "largest_pop") - - if (n_amb == 0L && n_nm == 0L && n_lp == 0L) return(invisible(NULL)) - - parts <- c( - if (n_amb > 0L) sprintf("%d ambiguous", n_amb), - if (n_nm > 0L) sprintf("%d with no match", n_nm), - if (n_lp > 0L) sprintf("%d used largest-population fallback", n_lp) - ) - - cli::cli_inform(c( - i = sprintf("Basket resolved %d of %d entries.", n_basket, n_input), - i = paste(parts, collapse = ", "), - i = "Inspect with `cog_basket_resolution(result)` or filter to problem rows with `cog_basket_unresolved(result)`." - )) - invisible(NULL) -} From 56fdd8e5b99d9e92629b7db7944ec8f1ec71ff35 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Tue, 28 Apr 2026 11:58:05 -0400 Subject: [PATCH 09/12] feat: export cog_basket_resolution() and cog_basket_unresolved() Sidecar accessors for basket-mode results. cog_basket_resolution() returns the full resolution tibble (drops candidates list-col by default for readable printing). cog_basket_unresolved() filters to ambiguous/no_match rows for iterative refinement. --- NAMESPACE | 2 ++ R/basket.R | 44 +++++++++++++++++++++++++++++++ man/cog_basket_resolution.Rd | 25 ++++++++++++++++++ man/cog_basket_unresolved.Rd | 20 ++++++++++++++ tests/testthat/test-basket.R | 51 ++++++++++++++++++++++++++++++++++++ 5 files changed, 142 insertions(+) create mode 100644 man/cog_basket_resolution.Rd create mode 100644 man/cog_basket_unresolved.Rd create mode 100644 tests/testthat/test-basket.R diff --git a/NAMESPACE b/NAMESPACE index 1550d74..518e06d 100644 --- a/NAMESPACE +++ b/NAMESPACE @@ -1,5 +1,7 @@ # Generated by roxygen2: do not edit by hand +export(cog_basket_resolution) +export(cog_basket_unresolved) export(cog_categories) export(cog_explain) export(cog_find_peers) diff --git a/R/basket.R b/R/basket.R index 0f9d91a..c6505cd 100644 --- a/R/basket.R +++ b/R/basket.R @@ -72,3 +72,47 @@ )) invisible(NULL) } + +#' Inspect basket-mode resolution sidecar +#' +#' Returns the resolution tibble attached to a basket-mode result of +#' [cog_gov_search()]. One row per input entry; `status` is one of +#' `"resolved"`, `"largest_pop"`, `"ambiguous"`, `"no_match"`. By default +#' the `candidates` list-column is dropped for readable printing; pass +#' `expand_candidates = TRUE` to keep it. +#' +#' @param x A tibble returned by basket-mode [cog_gov_search()]. +#' @param expand_candidates Logical. If `TRUE`, keeps the `candidates` +#' list-column (full-schema match candidates per input row). Default +#' `FALSE`. +#' @return A tibble with the resolution audit trail. +#' @export +cog_basket_resolution <- function(x, expand_candidates = FALSE) { + res <- attr(x, "resolution") + if (is.null(res)) { + cli::cli_abort(c( + "`x` has no resolution attribute.", + i = "Pass the result of basket-mode `cog_gov_search()` (length(name) > 1).", + i = "Single-name (utility) results do not carry a sidecar." + )) + } + if (!isTRUE(expand_candidates)) { + res$candidates <- NULL + } + res +} + +#' Filter a basket resolution to unresolved rows +#' +#' Convenience wrapper that returns just the rows where `status` is +#' `"ambiguous"` or `"no_match"` — the ones the user likely wants to +#' refine before piping into a query verb. The `candidates` list-column +#' is preserved so the user can drill into ambiguous match sets. +#' +#' @param x A tibble returned by basket-mode [cog_gov_search()]. +#' @return A tibble (subset of [cog_basket_resolution()]). +#' @export +cog_basket_unresolved <- function(x) { + res <- cog_basket_resolution(x, expand_candidates = TRUE) + res[res$status %in% c("ambiguous", "no_match"), , drop = FALSE] +} diff --git a/man/cog_basket_resolution.Rd b/man/cog_basket_resolution.Rd new file mode 100644 index 0000000..6ad2f24 --- /dev/null +++ b/man/cog_basket_resolution.Rd @@ -0,0 +1,25 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/basket.R +\name{cog_basket_resolution} +\alias{cog_basket_resolution} +\title{Inspect basket-mode resolution sidecar} +\usage{ +cog_basket_resolution(x, expand_candidates = FALSE) +} +\arguments{ +\item{x}{A tibble returned by basket-mode [cog_gov_search()].} + +\item{expand_candidates}{Logical. If `TRUE`, keeps the `candidates` +list-column (full-schema match candidates per input row). Default +`FALSE`.} +} +\value{ +A tibble with the resolution audit trail. +} +\description{ +Returns the resolution tibble attached to a basket-mode result of +[cog_gov_search()]. One row per input entry; `status` is one of +`"resolved"`, `"largest_pop"`, `"ambiguous"`, `"no_match"`. By default +the `candidates` list-column is dropped for readable printing; pass +`expand_candidates = TRUE` to keep it. +} diff --git a/man/cog_basket_unresolved.Rd b/man/cog_basket_unresolved.Rd new file mode 100644 index 0000000..727764c --- /dev/null +++ b/man/cog_basket_unresolved.Rd @@ -0,0 +1,20 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/basket.R +\name{cog_basket_unresolved} +\alias{cog_basket_unresolved} +\title{Filter a basket resolution to unresolved rows} +\usage{ +cog_basket_unresolved(x) +} +\arguments{ +\item{x}{A tibble returned by basket-mode [cog_gov_search()].} +} +\value{ +A tibble (subset of [cog_basket_resolution()]). +} +\description{ +Convenience wrapper that returns just the rows where `status` is +`"ambiguous"` or `"no_match"` — the ones the user likely wants to +refine before piping into a query verb. The `candidates` list-column +is preserved so the user can drill into ambiguous match sets. +} diff --git a/tests/testthat/test-basket.R b/tests/testthat/test-basket.R new file mode 100644 index 0000000..7ee1b4e --- /dev/null +++ b/tests/testthat/test-basket.R @@ -0,0 +1,51 @@ +test_that("cog_basket_resolution returns the sidecar tibble", { + basket <- suppressMessages(cog_gov_search( + name = c("Broward", "Notarealplace"), + state = c("FL", "NY") + )) + res <- cog_basket_resolution(basket) + expect_s3_class(res, "tbl_df") + expect_equal(nrow(res), 2L) + expect_setequal(colnames(res), c( + "query_name", "query_state", "query_type", "status", + "match_method", "canonical_govid", "gov_name", "n_candidates" + )) +}) + +test_that("cog_basket_resolution(expand_candidates = TRUE) keeps candidates list-col", { + basket <- suppressMessages(cog_gov_search( + name = c("Broward", "Notarealplace"), + state = c("FL", "NY") + )) + res <- cog_basket_resolution(basket, expand_candidates = TRUE) + expect_true("candidates" %in% colnames(res)) + expect_true(is.list(res$candidates)) +}) + +test_that("cog_basket_resolution errors on a non-basket tibble", { + utility <- cog_gov_search("BROWARD") + expect_error( + cog_basket_resolution(utility), + regexp = "no resolution attribute" + ) +}) + +test_that("cog_basket_unresolved filters to ambiguous and no_match", { + basket <- suppressMessages(cog_gov_search( + name = c("Broward", "San Diego", "Notarealplace"), + state = c("FL", "CA", "NY") + )) + unres <- cog_basket_unresolved(basket) + expect_equal(nrow(unres), 2L) + expect_setequal(unres$status, c("ambiguous", "no_match")) + expect_true("candidates" %in% colnames(unres)) +}) + +test_that("cog_basket_unresolved returns 0 rows when basket is clean", { + basket <- cog_gov_search( + name = c("BROWARD COUNTY", "SAN DIEGO CITY"), + state = c("FL", "CA") + ) + unres <- cog_basket_unresolved(basket) + expect_equal(nrow(unres), 0L) +}) From 7475696853897af1cdbd8034ee6d0775a55730cb Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Tue, 28 Apr 2026 12:10:28 -0400 Subject: [PATCH 10/12] test: smoke test cog_gov_search basket -> cog_spending pipe Confirms a basket result pipes cleanly into the existing query verb without any input-coercion friction. --- tests/testthat/test-spending.R | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/tests/testthat/test-spending.R b/tests/testthat/test-spending.R index 89cf731..3d39f19 100644 --- a/tests/testthat/test-spending.R +++ b/tests/testthat/test-spending.R @@ -116,3 +116,15 @@ test_that("cog_spending rejects data.frame without canonical_govid column", { bad <- tibble::tibble(foo = "bar") expect_error(cog_spending(bad, 2020L), "canonical_govid") }) + +test_that("cog_spending accepts a basket-mode cog_gov_search result", { + skip_if_no_corpus() + basket <- cog_gov_search( + name = c("BROWARD COUNTY", "SAN DIEGO COUNTY"), + state = c("FL", "CA") + ) + expect_equal(nrow(basket), 2L) + spending <- cog_spending(basket, years = 2019:2020, category = "Police") + expect_s3_class(spending, "tbl_df") + expect_setequal(unique(spending$canonical_govid), basket$canonical_govid) +}) From 24e67791a038029ce29d886d03e4cf8490482a57 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Tue, 28 Apr 2026 12:51:52 -0400 Subject: [PATCH 11/12] docs: roxygen, NEWS, and pkgdown for basket mode Adds full @description, @details (algorithm), @examples on cog_gov_search(); examples on cog_basket_resolution() / cog_basket_unresolved(); NEWS.md entry covering the new mode and the pattern->name rename; pkgdown reference entries for the two new exports. --- NEWS.md | 21 ++++++++ R/basket.R | 17 +++++++ R/search.R | 88 +++++++++++++++++++++++++++------- _pkgdown.yml | 6 +++ man/cog_basket_resolution.Rd | 10 ++++ man/cog_basket_unresolved.Rd | 9 ++++ man/cog_gov_search.Rd | 92 ++++++++++++++++++++++++++++++------ 7 files changed, 211 insertions(+), 32 deletions(-) create mode 100644 NEWS.md diff --git a/NEWS.md b/NEWS.md new file mode 100644 index 0000000..885398e --- /dev/null +++ b/NEWS.md @@ -0,0 +1,21 @@ +# uscogdata 0.1.0 (development) + +## New features + +* `cog_gov_search()` gains a **basket mode**: passing vector `name` + / `state` / `type` arguments resolves multiple place names in one + call and returns a tibble of canonical rows in input order, ready + to pipe into `cog_spending()` / `cog_revenue()`. Per-row resolution + follows an exact-then-substring matching algorithm with deterministic + disambiguation; ambiguous and missing entries are surfaced via a + sidecar audit tibble plus a single console summary message. +* New exports `cog_basket_resolution()` and `cog_basket_unresolved()` + expose the basket sidecar for iterative query refinement. + +## Breaking changes + +* The first formal of `cog_gov_search()` was renamed from `pattern` + to `name`. All existing call sites in `cog_explorer/` and the + package itself use positional first-arg, so this rename is + non-breaking in practice. Callers that pass `pattern = ...` by name + must update to `name = ...`. diff --git a/R/basket.R b/R/basket.R index c6505cd..bca6985 100644 --- a/R/basket.R +++ b/R/basket.R @@ -86,6 +86,15 @@ #' list-column (full-schema match candidates per input row). Default #' `FALSE`. #' @return A tibble with the resolution audit trail. +#' @examples +#' \dontrun{ +#' basket <- cog_gov_search( +#' name = c("Broward", "San Diego", "Notarealplace"), +#' state = c("FL", "CA", "NY") +#' ) +#' cog_basket_resolution(basket) +#' cog_basket_resolution(basket, expand_candidates = TRUE) +#' } #' @export cog_basket_resolution <- function(x, expand_candidates = FALSE) { res <- attr(x, "resolution") @@ -111,6 +120,14 @@ cog_basket_resolution <- function(x, expand_candidates = FALSE) { #' #' @param x A tibble returned by basket-mode [cog_gov_search()]. #' @return A tibble (subset of [cog_basket_resolution()]). +#' @examples +#' \dontrun{ +#' basket <- cog_gov_search( +#' name = c("Broward", "San Diego", "Notarealplace"), +#' state = c("FL", "CA", "NY") +#' ) +#' cog_basket_unresolved(basket) +#' } #' @export cog_basket_unresolved <- function(x) { res <- cog_basket_resolution(x, expand_candidates = TRUE) diff --git a/R/search.R b/R/search.R index 83f0c6c..b0b7da9 100644 --- a/R/search.R +++ b/R/search.R @@ -2,29 +2,81 @@ #' Search for governments by name, state, and/or type #' -#' Two modes: +#' Resolves human-readable place names into rows of `canonical_fips_xwalk`, +#' the cross-vintage canonical-government registry. Operates in two modes: #' -#' * **Utility mode** (single `name`): returns all rows from -#' `canonical_fips_xwalk` whose `gov_name` matches the regex -#' case-insensitively, sorted by `population_acs` descending. +#' * **Utility mode** (single `name`, the original behavior): returns all +#' rows whose `gov_name` matches the regex case-insensitively, sorted by +#' `population_acs` descending. Useful for exploratory lookups. #' * **Basket mode** (`length(name) > 1`): resolves each input row to a -#' single canonical govid via exact-then-substring matching with -#' deterministic disambiguation. Returns up to `length(name)` rows in -#' input order plus a `"resolution"` sidecar attribute. See -#' [cog_basket_resolution()]. +#' single canonical govid and returns a tibble in input order, suitable +#' for piping straight into [cog_spending()] / [cog_revenue()] / +#' [cog_geographic_rollup()]. Carries an audit sidecar accessible via +#' [cog_basket_resolution()] / [cog_basket_unresolved()]. +#' +#' @details +#' **Basket-mode resolution algorithm** (per input row): +#' 1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`. +#' 2. **Exact pass:** case-insensitive equality against `gov_name`. +#' Single hit -> resolved. Multiple -> step 4. +#' 3. **Substring fallback:** case-insensitive regex against `gov_name`. +#' Single hit -> resolved (`match_method = "substring"`). Zero hits -> +#' `status = "no_match"`. Multiple hits -> step 4. +#' 4. **Disambiguation:** if matches share one `govs_type`, pick the +#' largest-population row (`status = "largest_pop"`). If they span >=2 +#' types, no row is added (`status = "ambiguous"`); the user should +#' re-run with `type` specified. +#' +#' Resolved rows form the returned tibble in input order. Unresolved +#' inputs (`ambiguous` / `no_match`) appear only in the sidecar. #' #' @param name Character vector of place name(s). Length 1 = utility mode; #' length >1 = basket mode. -#' @param state Either a 2-letter USPS abbreviation, a FIPS integer, or -#' `NULL`. Length 1 recycles across all entries in basket mode. -#' @param type Government type: an integer in `0:3` or one of `"state"`, -#' `"county"`, `"city"`, `"township"`, or `NA` (per-row optional in -#' basket mode). Passing `4`, `5`, `"special_district"`, or -#' `"school_district"` emits an explanatory message and returns an -#' empty tibble (v0.1 corpus excludes those types). -#' @return Tibble from `canonical_fips_xwalk`. In utility mode, sorted by -#' `population_acs` descending (`NULL`s last). In basket mode, in input -#' order, with `attr(result, "resolution")` set to the sidecar tibble. +#' @param state 2-letter USPS abbreviation (e.g. `"FL"`), FIPS integer +#' (e.g. `12`), or `NULL`. In basket mode, length 1 recycles across +#' all entries; otherwise must match `length(name)`. +#' @param type Government type: integer in `0:3` or one of `"state"`, +#' `"county"`, `"city"`, `"township"`, or `NA`/`NULL`. Per-row optional +#' in basket mode (recycles from length 1). Excluded types `4`/`5` (or +#' `"special_district"` / `"school_district"`) trigger an explanatory +#' message and an empty result. +#' @return A tibble of `canonical_fips_xwalk` rows. In utility mode, all +#' matches sorted by `population_acs` desc. In basket mode, resolved +#' rows in input order, with `attr(., "resolution")` set to the +#' sidecar tibble. +#' @seealso [cog_basket_resolution()], [cog_basket_unresolved()], +#' [cog_spending()], [cog_revenue()]. +#' @examples +#' \dontrun{ +#' # Utility mode — exploratory regex lookup +#' cog_gov_search("broward", state = "FL") +#' +#' # Basket mode — resolve a known cohort +#' basket <- cog_gov_search( +#' name = c("BROWARD COUNTY", "SAN DIEGO CITY", "AUSTIN CITY"), +#' state = c("FL", "CA", "TX") +#' ) +#' basket +#' +#' # Inspect resolution audit +#' cog_basket_resolution(basket) +#' +#' # Pipe into a spending query +#' library(dplyr) +#' basket |> cog_spending(years = 2019:2020, category = "Police") +#' +#' # Iteratively refine ambiguous matches +#' partial <- cog_gov_search( +#' name = c("Broward", "San Diego"), # San Diego is ambiguous +#' state = c("FL", "CA") +#' ) +#' cog_basket_unresolved(partial) +#' refined <- cog_gov_search( +#' name = c("Broward", "San Diego"), +#' state = c("FL", "CA"), +#' type = c(NA, "city") # disambiguate +#' ) +#' } #' @export cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { if (!is.null(type) && length(type) == 1L && .is_excluded_type(type)) { diff --git a/_pkgdown.yml b/_pkgdown.yml index 3aeefe7..2022548 100644 --- a/_pkgdown.yml +++ b/_pkgdown.yml @@ -3,6 +3,12 @@ template: bootstrap: 5 reference: + - title: Search & basket + desc: Resolve place names into canonical govids. + contents: + - cog_gov_search + - cog_basket_resolution + - cog_basket_unresolved - title: Session contents: - has_keyword("internal") diff --git a/man/cog_basket_resolution.Rd b/man/cog_basket_resolution.Rd index 6ad2f24..3c1825f 100644 --- a/man/cog_basket_resolution.Rd +++ b/man/cog_basket_resolution.Rd @@ -23,3 +23,13 @@ Returns the resolution tibble attached to a basket-mode result of the `candidates` list-column is dropped for readable printing; pass `expand_candidates = TRUE` to keep it. } +\examples{ +\dontrun{ +basket <- cog_gov_search( + name = c("Broward", "San Diego", "Notarealplace"), + state = c("FL", "CA", "NY") +) +cog_basket_resolution(basket) +cog_basket_resolution(basket, expand_candidates = TRUE) +} +} diff --git a/man/cog_basket_unresolved.Rd b/man/cog_basket_unresolved.Rd index 727764c..e6a5145 100644 --- a/man/cog_basket_unresolved.Rd +++ b/man/cog_basket_unresolved.Rd @@ -18,3 +18,12 @@ Convenience wrapper that returns just the rows where `status` is refine before piping into a query verb. The `candidates` list-column is preserved so the user can drill into ambiguous match sets. } +\examples{ +\dontrun{ +basket <- cog_gov_search( + name = c("Broward", "San Diego", "Notarealplace"), + state = c("FL", "CA", "NY") +) +cog_basket_unresolved(basket) +} +} diff --git a/man/cog_gov_search.Rd b/man/cog_gov_search.Rd index 0d2a0e6..4a6b55b 100644 --- a/man/cog_gov_search.Rd +++ b/man/cog_gov_search.Rd @@ -7,24 +7,88 @@ cog_gov_search(name = NULL, state = NULL, type = NULL) } \arguments{ -\item{name}{Character regex matched case-insensitively against -`gov_name`. `NULL` (default) means no name filter.} +\item{name}{Character vector of place name(s). Length 1 = utility mode; +length >1 = basket mode.} -\item{state}{Either a 2-letter USPS abbreviation (e.g. `"FL"`), a FIPS -integer (e.g. `12`), or `NULL`.} +\item{state}{2-letter USPS abbreviation (e.g. `"FL"`), FIPS integer +(e.g. `12`), or `NULL`. In basket mode, length 1 recycles across +all entries; otherwise must match `length(name)`.} -\item{type}{Government type: an integer in `0:3` or one of `"state"`, -`"county"`, `"city"`, `"township"`. Passing `4`, `5`, -`"special_district"`, or `"school_district"` emits an explanatory -message and returns an empty tibble (v0.1 corpus excludes those types).} +\item{type}{Government type: integer in `0:3` or one of `"state"`, +`"county"`, `"city"`, `"township"`, or `NA`/`NULL`. Per-row optional +in basket mode (recycles from length 1). Excluded types `4`/`5` (or +`"special_district"` / `"school_district"`) trigger an explanatory +message and an empty result.} } \value{ -Tibble from `canonical_fips_xwalk` sorted by `population_acs` - descending (`NULL`s last). +A tibble of `canonical_fips_xwalk` rows. In utility mode, all + matches sorted by `population_acs` desc. In basket mode, resolved + rows in input order, with `attr(., "resolution")` set to the + sidecar tibble. } \description{ -Returns rows from `canonical_fips_xwalk` matching the supplied filters. -Intended as the entry point users call to resolve a human-readable place -name into one or more `canonical_govid` values before calling -[cog_spending()] / [cog_revenue()] / etc. +Resolves human-readable place names into rows of `canonical_fips_xwalk`, +the cross-vintage canonical-government registry. Operates in two modes: +} +\details{ +* **Utility mode** (single `name`, the original behavior): returns all + rows whose `gov_name` matches the regex case-insensitively, sorted by + `population_acs` descending. Useful for exploratory lookups. +* **Basket mode** (`length(name) > 1`): resolves each input row to a + single canonical govid and returns a tibble in input order, suitable + for piping straight into [cog_spending()] / [cog_revenue()] / + [cog_geographic_rollup()]. Carries an audit sidecar accessible via + [cog_basket_resolution()] / [cog_basket_unresolved()]. + + +**Basket-mode resolution algorithm** (per input row): +1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`. +2. **Exact pass:** case-insensitive equality against `gov_name`. + Single hit -> resolved. Multiple -> step 4. +3. **Substring fallback:** case-insensitive regex against `gov_name`. + Single hit -> resolved (`match_method = "substring"`). Zero hits -> + `status = "no_match"`. Multiple hits -> step 4. +4. **Disambiguation:** if matches share one `govs_type`, pick the + largest-population row (`status = "largest_pop"`). If they span >=2 + types, no row is added (`status = "ambiguous"`); the user should + re-run with `type` specified. + +Resolved rows form the returned tibble in input order. Unresolved +inputs (`ambiguous` / `no_match`) appear only in the sidecar. +} +\examples{ +\dontrun{ +# Utility mode — exploratory regex lookup +cog_gov_search("broward", state = "FL") + +# Basket mode — resolve a known cohort +basket <- cog_gov_search( + name = c("BROWARD COUNTY", "SAN DIEGO CITY", "AUSTIN CITY"), + state = c("FL", "CA", "TX") +) +basket + +# Inspect resolution audit +cog_basket_resolution(basket) + +# Pipe into a spending query +library(dplyr) +basket |> cog_spending(years = 2019:2020, category = "Police") + +# Iteratively refine ambiguous matches +partial <- cog_gov_search( + name = c("Broward", "San Diego"), # San Diego is ambiguous + state = c("FL", "CA") +) +cog_basket_unresolved(partial) +refined <- cog_gov_search( + name = c("Broward", "San Diego"), + state = c("FL", "CA"), + type = c(NA, "city") # disambiguate +) +} +} +\seealso{ +[cog_basket_resolution()], [cog_basket_unresolved()], + [cog_spending()], [cog_revenue()]. } From efc0bd16b1e16877e40fa2e61d628d9c772db390 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Tue, 28 Apr 2026 14:16:37 -0400 Subject: [PATCH 12/12] fix(search): soft-fail on per-row excluded type and malformed regex name Cross-task review found two edge cases that violated the basket-mode soft-fail contract: - Per-row excluded type (e.g. type = c(NA, "special_district")) hit .coerce_type()'s abort inside the per-row resolver, killing the whole basket call. Now treated as no_match in the sidecar. - Malformed regex in the substring fallback (e.g. name = "San(Diego") propagated DuckDB engine errors. .escape_regex() now backslash- escapes meta characters before the regexp_matches call. Utility- mode regex behavior is unchanged. Plus a new public-surface test for the all-no-match case. --- R/basket.R | 5 ++- R/search.R | 23 +++++++++++++- tests/testthat/test-search.R | 60 ++++++++++++++++++++++++++++++++++++ 3 files changed, 86 insertions(+), 2 deletions(-) diff --git a/R/basket.R b/R/basket.R index bca6985..9f737e4 100644 --- a/R/basket.R +++ b/R/basket.R @@ -39,9 +39,12 @@ } # Convert a type input (integer-like or label) into the canonical label -# string used in the sidecar query_type column. +# string used in the sidecar query_type column. Excluded types (4/5 / +# special_district / school_district) are returned as-is so the sidecar +# records what the user passed without calling .coerce_type() (which aborts). #' @noRd .type_to_label <- function(type) { + if (.is_excluded_type(type)) return(as.character(type)) int_type <- .coerce_type(type) unname(c("0" = "state", "1" = "county", "2" = "city", "3" = "township")[[as.character(int_type)]]) } diff --git a/R/search.R b/R/search.R index b0b7da9..d2b78d6 100644 --- a/R/search.R +++ b/R/search.R @@ -132,6 +132,14 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { ) } +#' @noRd +.escape_regex <- function(x) { + # Backslash-escape POSIX regex metacharacters so `name` is treated as a + # literal substring in the DuckDB regexp_matches call (substring fallback + # only; utility-mode intentionally preserves regex behavior). + gsub("([\\^$.|?*+(){}\\[\\]])", "\\\\\\1", x, perl = TRUE) +} + #' @noRd .is_excluded_type <- function(type) { excluded <- c("4", "5", "special_district", "school_district") @@ -245,6 +253,19 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { )) } + # Short-circuit: excluded type (4/5 / special_district / school_district) + # -> no_match without SQL, preserving soft-fail contract. + if (!is.na(type) && .is_excluded_type(type)) { + empty <- .empty_xwalk_tibble() + return(list( + status = "no_match", + match_method = NA_character_, + n_candidates = 0L, + row = empty, + candidates = empty + )) + } + preds <- character(0) if (!is.na(state)) { st_fips <- .coerce_state_to_fips(state) @@ -282,7 +303,7 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { "SELECT * FROM canonical_fips_xwalk", base_where, conj, - sprintf("regexp_matches(gov_name, %s, 'i')", .sql_lit_chr(name)) + sprintf("regexp_matches(gov_name, %s, 'i')", .sql_lit_chr(.escape_regex(name))) ) sub <- tibble::as_tibble(DBI::dbGetQuery(con, sub_sql)) diff --git a/tests/testthat/test-search.R b/tests/testthat/test-search.R index ef2a892..07bdb7e 100644 --- a/tests/testthat/test-search.R +++ b/tests/testthat/test-search.R @@ -375,3 +375,63 @@ test_that("cog_gov_search basket mode message points to the sidecar accessor", { regexp = "cog_basket_resolution" ) }) + +# ---- F1: per-row excluded type soft-fail ---- + +test_that(".resolve_basket_row treats excluded type as no_match (not abort)", { + con <- uscogdata:::.ensure_session() + out <- uscogdata:::.resolve_basket_row( + name = "Some District", state = "CA", type = "special_district", con = con + ) + expect_equal(out$status, "no_match") + expect_true(is.na(out$match_method)) + expect_equal(out$n_candidates, 0L) +}) + +test_that("cog_gov_search basket mode skips per-row excluded type without aborting", { + basket <- suppressMessages(cog_gov_search( + name = c("BROWARD COUNTY", "Some District"), + state = c("FL", "FL"), + type = c(NA, "special_district") + )) + # Broward should resolve; the special_district row should be no_match. + expect_equal(nrow(basket), 1L) + expect_equal(basket$canonical_govid, "101006006") + res <- attr(basket, "resolution") + expect_equal(res$status, c("resolved", "no_match")) + # query_type should record what the user passed for the excluded-type row + expect_equal(res$query_type, c(NA_character_, "special_district")) +}) + +# ---- F2: malformed regex name soft-fail ---- + +test_that(".resolve_basket_row treats malformed regex name as no_match", { + con <- uscogdata:::.ensure_session() + # Unbalanced parens would be a regex parse error if not escaped. + out <- uscogdata:::.resolve_basket_row( + name = "San(Diego", state = "CA", type = NA_character_, con = con + ) + expect_equal(out$status, "no_match") +}) + +test_that(".resolve_basket_row escapes regex metacharacters in name", { + con <- uscogdata:::.ensure_session() + # Confirm that names with various metacharacters don't error. + expect_no_error(uscogdata:::.resolve_basket_row( + name = "Foo*Bar+Baz", state = "FL", type = NA_character_, con = con + )) +}) + +# ---- F3: all-no-match basket public surface ---- + +test_that("cog_gov_search basket all-no-match returns 0-row tibble with full sidecar", { + basket <- suppressMessages(cog_gov_search( + name = c("Notarealplace1", "Notarealplace2"), + state = c("NY", "CA") + )) + expect_equal(nrow(basket), 0L) + expect_true("canonical_govid" %in% names(basket)) + res <- attr(basket, "resolution") + expect_equal(nrow(res), 2L) + expect_true(all(res$status == "no_match")) +})