Merge pull request 'feat(search): basket mode for cog_gov_search()' (#1) from feat/cog-gov-search-basket-mode into main
R-CMD-check / check (push) Successful in 1m34s

This commit was merged in pull request #1.
This commit is contained in:
2026-04-28 15:09:44 -04:00
11 changed files with 1015 additions and 35 deletions
+2
View File
@@ -1,5 +1,7 @@
# Generated by roxygen2: do not edit by hand # Generated by roxygen2: do not edit by hand
export(cog_basket_resolution)
export(cog_basket_unresolved)
export(cog_categories) export(cog_categories)
export(cog_explain) export(cog_explain)
export(cog_find_peers) export(cog_find_peers)
+21
View File
@@ -0,0 +1,21 @@
# uscogdata 0.1.0 (development)
## New features
* `cog_gov_search()` gains a **basket mode**: passing vector `name`
/ `state` / `type` arguments resolves multiple place names in one
call and returns a tibble of canonical rows in input order, ready
to pipe into `cog_spending()` / `cog_revenue()`. Per-row resolution
follows an exact-then-substring matching algorithm with deterministic
disambiguation; ambiguous and missing entries are surfaced via a
sidecar audit tibble plus a single console summary message.
* New exports `cog_basket_resolution()` and `cog_basket_unresolved()`
expose the basket sidecar for iterative query refinement.
## Breaking changes
* The first formal of `cog_gov_search()` was renamed from `pattern`
to `name`. All existing call sites in `cog_explorer/` and the
package itself use positional first-arg, so this rename is
non-breaking in practice. Callers that pass `pattern = ...` by name
must update to `name = ...`.
+138
View File
@@ -0,0 +1,138 @@
# R/basket.R
#
# Internals supporting the basket-mode sidecar (constructed in
# .resolve_basket() — see R/search.R) plus the user-facing accessors
# cog_basket_resolution() and cog_basket_unresolved() (added in a
# later step).
# Build the sidecar tibble. One row per input; carries query_*, status,
# match_method, canonical_govid, gov_name, n_candidates, and a list-col
# `candidates` of full-schema match-candidate tibbles.
#' @noRd
.build_sidecar <- function(args, resolved) {
type_label <- unname(vapply(args$type, function(t) {
if (is.na(t)) NA_character_ else .type_to_label(t)
}, character(1)))
status <- vapply(resolved, `[[`, character(1), "status")
method <- vapply(resolved, `[[`, character(1), "match_method")
ncand <- vapply(resolved, `[[`, integer(1), "n_candidates")
govid <- vapply(resolved, function(r) {
if (nrow(r$row) == 0L) NA_character_ else r$row$canonical_govid[1L]
}, character(1))
gname <- vapply(resolved, function(r) {
if (nrow(r$row) == 0L) NA_character_ else r$row$gov_name[1L]
}, character(1))
cands <- lapply(resolved, `[[`, "candidates")
tibble::tibble(
query_name = args$name,
query_state = args$state,
query_type = type_label,
status = status,
match_method = method,
canonical_govid = govid,
gov_name = gname,
n_candidates = ncand,
candidates = cands
)
}
# Convert a type input (integer-like or label) into the canonical label
# string used in the sidecar query_type column. Excluded types (4/5 /
# special_district / school_district) are returned as-is so the sidecar
# records what the user passed without calling .coerce_type() (which aborts).
#' @noRd
.type_to_label <- function(type) {
if (.is_excluded_type(type)) return(as.character(type))
int_type <- .coerce_type(type)
unname(c("0" = "state", "1" = "county", "2" = "city", "3" = "township")[[as.character(int_type)]])
}
# Single post-resolution summary message. Silent on clean baskets;
# emits one cli_inform with two-line body otherwise.
#' @noRd
.basket_summary_message <- function(sidecar) {
status <- sidecar$status
n_input <- length(status)
n_basket <- sum(status %in% c("resolved", "largest_pop"))
n_amb <- sum(status == "ambiguous")
n_nm <- sum(status == "no_match")
n_lp <- sum(status == "largest_pop")
if (n_amb == 0L && n_nm == 0L && n_lp == 0L) return(invisible(NULL))
parts <- c(
if (n_amb > 0L) sprintf("%d ambiguous", n_amb),
if (n_nm > 0L) sprintf("%d with no match", n_nm),
if (n_lp > 0L) sprintf("%d used largest-population fallback", n_lp)
)
cli::cli_inform(c(
i = sprintf("Basket resolved %d of %d entries.", n_basket, n_input),
i = paste(parts, collapse = ", "),
i = "Inspect with `cog_basket_resolution(result)` or filter to problem rows with `cog_basket_unresolved(result)`."
))
invisible(NULL)
}
#' Inspect basket-mode resolution sidecar
#'
#' Returns the resolution tibble attached to a basket-mode result of
#' [cog_gov_search()]. One row per input entry; `status` is one of
#' `"resolved"`, `"largest_pop"`, `"ambiguous"`, `"no_match"`. By default
#' the `candidates` list-column is dropped for readable printing; pass
#' `expand_candidates = TRUE` to keep it.
#'
#' @param x A tibble returned by basket-mode [cog_gov_search()].
#' @param expand_candidates Logical. If `TRUE`, keeps the `candidates`
#' list-column (full-schema match candidates per input row). Default
#' `FALSE`.
#' @return A tibble with the resolution audit trail.
#' @examples
#' \dontrun{
#' basket <- cog_gov_search(
#' name = c("Broward", "San Diego", "Notarealplace"),
#' state = c("FL", "CA", "NY")
#' )
#' cog_basket_resolution(basket)
#' cog_basket_resolution(basket, expand_candidates = TRUE)
#' }
#' @export
cog_basket_resolution <- function(x, expand_candidates = FALSE) {
res <- attr(x, "resolution")
if (is.null(res)) {
cli::cli_abort(c(
"`x` has no resolution attribute.",
i = "Pass the result of basket-mode `cog_gov_search()` (length(name) > 1).",
i = "Single-name (utility) results do not carry a sidecar."
))
}
if (!isTRUE(expand_candidates)) {
res$candidates <- NULL
}
res
}
#' Filter a basket resolution to unresolved rows
#'
#' Convenience wrapper that returns just the rows where `status` is
#' `"ambiguous"` or `"no_match"` — the ones the user likely wants to
#' refine before piping into a query verb. The `candidates` list-column
#' is preserved so the user can drill into ambiguous match sets.
#'
#' @param x A tibble returned by basket-mode [cog_gov_search()].
#' @return A tibble (subset of [cog_basket_resolution()]).
#' @examples
#' \dontrun{
#' basket <- cog_gov_search(
#' name = c("Broward", "San Diego", "Notarealplace"),
#' state = c("FL", "CA", "NY")
#' )
#' cog_basket_unresolved(basket)
#' }
#' @export
cog_basket_unresolved <- function(x) {
res <- cog_basket_resolution(x, expand_candidates = TRUE)
res[res$status %in% c("ambiguous", "no_match"), , drop = FALSE]
}
+279 -20
View File
@@ -2,40 +2,105 @@
#' Search for governments by name, state, and/or type #' Search for governments by name, state, and/or type
#' #'
#' Returns rows from `canonical_fips_xwalk` matching the supplied filters. #' Resolves human-readable place names into rows of `canonical_fips_xwalk`,
#' Intended as the entry point users call to resolve a human-readable place #' the cross-vintage canonical-government registry. Operates in two modes:
#' name into one or more `canonical_govid` values before calling
#' [cog_spending()] / [cog_revenue()] / etc.
#' #'
#' @param pattern Character regex matched case-insensitively against #' * **Utility mode** (single `name`, the original behavior): returns all
#' `gov_name`. `NULL` (default) means no name filter. #' rows whose `gov_name` matches the regex case-insensitively, sorted by
#' @param state Either a 2-letter USPS abbreviation (e.g. `"FL"`), a FIPS #' `population_acs` descending. Useful for exploratory lookups.
#' integer (e.g. `12`), or `NULL`. #' * **Basket mode** (`length(name) > 1`): resolves each input row to a
#' @param type Government type: an integer in `0:3` or one of `"state"`, #' single canonical govid and returns a tibble in input order, suitable
#' `"county"`, `"city"`, `"township"`. Passing `4`, `5`, #' for piping straight into [cog_spending()] / [cog_revenue()] /
#' `"special_district"`, or `"school_district"` emits an explanatory #' [cog_geographic_rollup()]. Carries an audit sidecar accessible via
#' message and returns an empty tibble (v0.1 corpus excludes those types). #' [cog_basket_resolution()] / [cog_basket_unresolved()].
#' @return Tibble from `canonical_fips_xwalk` sorted by `population_acs` #'
#' descending (`NULL`s last). #' @details
#' **Basket-mode resolution algorithm** (per input row):
#' 1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`.
#' 2. **Exact pass:** case-insensitive equality against `gov_name`.
#' Single hit -> resolved. Multiple -> step 4.
#' 3. **Substring fallback:** case-insensitive regex against `gov_name`.
#' Single hit -> resolved (`match_method = "substring"`). Zero hits ->
#' `status = "no_match"`. Multiple hits -> step 4.
#' 4. **Disambiguation:** if matches share one `govs_type`, pick the
#' largest-population row (`status = "largest_pop"`). If they span >=2
#' types, no row is added (`status = "ambiguous"`); the user should
#' re-run with `type` specified.
#'
#' Resolved rows form the returned tibble in input order. Unresolved
#' inputs (`ambiguous` / `no_match`) appear only in the sidecar.
#'
#' @param name Character vector of place name(s). Length 1 = utility mode;
#' length >1 = basket mode.
#' @param state 2-letter USPS abbreviation (e.g. `"FL"`), FIPS integer
#' (e.g. `12`), or `NULL`. In basket mode, length 1 recycles across
#' all entries; otherwise must match `length(name)`.
#' @param type Government type: integer in `0:3` or one of `"state"`,
#' `"county"`, `"city"`, `"township"`, or `NA`/`NULL`. Per-row optional
#' in basket mode (recycles from length 1). Excluded types `4`/`5` (or
#' `"special_district"` / `"school_district"`) trigger an explanatory
#' message and an empty result.
#' @return A tibble of `canonical_fips_xwalk` rows. In utility mode, all
#' matches sorted by `population_acs` desc. In basket mode, resolved
#' rows in input order, with `attr(., "resolution")` set to the
#' sidecar tibble.
#' @seealso [cog_basket_resolution()], [cog_basket_unresolved()],
#' [cog_spending()], [cog_revenue()].
#' @examples
#' \dontrun{
#' # Utility mode — exploratory regex lookup
#' cog_gov_search("broward", state = "FL")
#'
#' # Basket mode — resolve a known cohort
#' basket <- cog_gov_search(
#' name = c("BROWARD COUNTY", "SAN DIEGO CITY", "AUSTIN CITY"),
#' state = c("FL", "CA", "TX")
#' )
#' basket
#'
#' # Inspect resolution audit
#' cog_basket_resolution(basket)
#'
#' # Pipe into a spending query
#' library(dplyr)
#' basket |> cog_spending(years = 2019:2020, category = "Police")
#'
#' # Iteratively refine ambiguous matches
#' partial <- cog_gov_search(
#' name = c("Broward", "San Diego"), # San Diego is ambiguous
#' state = c("FL", "CA")
#' )
#' cog_basket_unresolved(partial)
#' refined <- cog_gov_search(
#' name = c("Broward", "San Diego"),
#' state = c("FL", "CA"),
#' type = c(NA, "city") # disambiguate
#' )
#' }
#' @export #' @export
cog_gov_search <- function(pattern = NULL, state = NULL, type = NULL) { cog_gov_search <- function(name = NULL, state = NULL, type = NULL) {
if (!is.null(type) && .is_excluded_type(type)) { if (!is.null(type) && length(type) == 1L && .is_excluded_type(type)) {
cli::cli_inform(c( cli::cli_inform(c(
i = "v0.1 covers gov_types 0-3 (state/county/city/township) only.", i = "v0.1 covers gov_types 0-3 (state/county/city/township) only.",
i = "Types 4 (special districts) and 5 (school districts) are excluded; see vignette('coverage-scope')." i = "Types 4 (special districts) and 5 (school districts) are excluded; see vignette('coverage-scope')."
)) ))
return(.empty_xwalk_tibble()) return(.empty_xwalk_tibble())
} }
con <- .ensure_session() con <- .ensure_session()
if (length(name) > 1L) {
return(.resolve_basket(name = name, state = state, type = type, con = con))
}
preds <- character(0) preds <- character(0)
if (!is.null(pattern)) { if (!is.null(name)) {
if (!is.character(pattern) || length(pattern) != 1L) { if (!is.character(name) || length(name) != 1L) {
cli::cli_abort("`pattern` must be a length-1 character string.") cli::cli_abort("`name` must be a length-1 character string.")
} }
preds <- c(preds, preds <- c(preds,
sprintf("regexp_matches(gov_name, %s, 'i')", sprintf("regexp_matches(gov_name, %s, 'i')",
.sql_lit_chr(pattern))) .sql_lit_chr(name)))
} }
if (!is.null(state)) { if (!is.null(state)) {
st_fips <- .coerce_state_to_fips(state) st_fips <- .coerce_state_to_fips(state)
@@ -67,6 +132,14 @@ cog_gov_search <- function(pattern = NULL, state = NULL, type = NULL) {
) )
} }
#' @noRd
.escape_regex <- function(x) {
# Backslash-escape POSIX regex metacharacters so `name` is treated as a
# literal substring in the DuckDB regexp_matches call (substring fallback
# only; utility-mode intentionally preserves regex behavior).
gsub("([\\^$.|?*+(){}\\[\\]])", "\\\\\\1", x, perl = TRUE)
}
#' @noRd #' @noRd
.is_excluded_type <- function(type) { .is_excluded_type <- function(type) {
excluded <- c("4", "5", "special_district", "school_district") excluded <- c("4", "5", "special_district", "school_district")
@@ -120,3 +193,189 @@ cog_gov_search <- function(pattern = NULL, state = NULL, type = NULL) {
WV = "54", WI = "55", WY = "56", WV = "54", WI = "55", WY = "56",
AS = "60", GU = "66", MP = "69", PR = "72", VI = "78" AS = "60", GU = "66", MP = "69", PR = "72", VI = "78"
) )
# Validate basket-mode inputs. Returns a list with normalized character
# vectors `name`, `state`, `type`, all of length n = length(name).
# `state` and `type` of length 1 are recycled; lengths must be 1 or n
# otherwise. NULL state/type become a vector of NA_character_.
#' @noRd
.validate_basket_args <- function(name, state, type) {
if (!is.character(name)) {
cli::cli_abort("`name` must be a character vector.")
}
n <- length(name)
state_norm <- if (is.null(state)) {
rep(NA_character_, n)
} else if (length(state) == 1L) {
rep(as.character(state), n)
} else if (length(state) == n) {
as.character(state)
} else {
cli::cli_abort(
"`state` must be length 1 or {n} (length of `name`); got {length(state)}."
)
}
type_norm <- if (is.null(type)) {
rep(NA_character_, n)
} else if (length(type) == 1L) {
rep(as.character(type), n)
} else if (length(type) == n) {
as.character(type)
} else {
cli::cli_abort(
"`type` must be length 1 or {n} (length of `name`); got {length(type)}."
)
}
list(name = name, state = state_norm, type = type_norm)
}
# Resolve a single basket-mode input row. Returns a list with components:
# status : "resolved" | "largest_pop" | "ambiguous" | "no_match"
# match_method : "exact" | "substring" | NA_character_
# n_candidates : int
# row : tibble (single resolved row, or 0-row tibble for unresolved)
# candidates : tibble (all rows that matched, for sidecar)
# Internal use only; takes an active DuckDB connection to reuse the session.
#' @noRd
.resolve_basket_row <- function(name, state, type, con) {
# Short-circuit: empty/whitespace name -> no_match without SQL.
if (!nzchar(trimws(name))) {
empty <- .empty_xwalk_tibble()
return(list(
status = "no_match",
match_method = NA_character_,
n_candidates = 0L,
row = empty,
candidates = empty
))
}
# Short-circuit: excluded type (4/5 / special_district / school_district)
# -> no_match without SQL, preserving soft-fail contract.
if (!is.na(type) && .is_excluded_type(type)) {
empty <- .empty_xwalk_tibble()
return(list(
status = "no_match",
match_method = NA_character_,
n_candidates = 0L,
row = empty,
candidates = empty
))
}
preds <- character(0)
if (!is.na(state)) {
st_fips <- .coerce_state_to_fips(state)
preds <- c(preds, sprintf("fips_state = %s", .sql_lit_chr(st_fips)))
}
if (!is.na(type)) {
int_type <- .coerce_type(type)
preds <- c(preds, sprintf("govs_type = %d", int_type))
}
base_where <- if (length(preds) == 0L) "" else paste("WHERE", paste(preds, collapse = " AND "))
conj <- if (nzchar(base_where)) "AND" else "WHERE"
exact_sql <- paste(
"SELECT * FROM canonical_fips_xwalk",
base_where,
conj,
sprintf("LOWER(gov_name) = LOWER(%s)", .sql_lit_chr(name))
)
exact <- tibble::as_tibble(DBI::dbGetQuery(con, exact_sql))
if (nrow(exact) == 1L) {
return(list(
status = "resolved",
match_method = "exact",
n_candidates = 1L,
row = exact,
candidates = exact
))
}
if (nrow(exact) > 1L) {
return(.disambiguate(exact, method = "exact"))
}
sub_sql <- paste(
"SELECT * FROM canonical_fips_xwalk",
base_where,
conj,
sprintf("regexp_matches(gov_name, %s, 'i')", .sql_lit_chr(.escape_regex(name)))
)
sub <- tibble::as_tibble(DBI::dbGetQuery(con, sub_sql))
if (nrow(sub) == 0L) {
return(list(
status = "no_match",
match_method = NA_character_,
n_candidates = 0L,
row = sub,
candidates = sub
))
}
if (nrow(sub) == 1L) {
return(list(
status = "resolved",
match_method = "substring",
n_candidates = 1L,
row = sub,
candidates = sub
))
}
.disambiguate(sub, method = "substring")
}
# Disambiguate a multi-row match set. Either picks the largest-pop row
# (within single-type) or returns an ambiguous result with no basket row.
#' @noRd
.disambiguate <- function(matches, method) {
types <- unique(matches$govs_type)
if (length(types) == 1L) {
pick <- matches[order(-matches$population_acs, na.last = TRUE), , drop = FALSE][1L, , drop = FALSE]
return(list(
status = "largest_pop",
match_method = method,
n_candidates = nrow(matches),
row = pick,
candidates = matches
))
}
empty <- matches[0, , drop = FALSE]
list(
status = "ambiguous",
match_method = NA_character_,
n_candidates = nrow(matches),
row = empty,
candidates = matches
)
}
# Orchestrates basket-mode resolution: validate, per-row resolve,
# assemble the basket tibble + sidecar, attach the sidecar as an attr.
#' @noRd
.resolve_basket <- function(name, state, type, con) {
args <- .validate_basket_args(name = name, state = state, type = type)
n <- length(args$name)
resolved <- vector("list", n)
for (i in seq_len(n)) {
resolved[[i]] <- .resolve_basket_row(
name = args$name[i],
state = args$state[i],
type = args$type[i],
con = con
)
}
basket_rows <- lapply(resolved, function(r) r$row)
basket <- dplyr::bind_rows(basket_rows[vapply(basket_rows, function(r) nrow(r) > 0L, logical(1))])
if (nrow(basket) == 0L) basket <- .empty_xwalk_tibble()
sidecar <- .build_sidecar(args, resolved)
attr(basket, "resolution") <- sidecar
.basket_summary_message(sidecar)
basket
}
+6
View File
@@ -3,6 +3,12 @@ template:
bootstrap: 5 bootstrap: 5
reference: reference:
- title: Search & basket
desc: Resolve place names into canonical govids.
contents:
- cog_gov_search
- cog_basket_resolution
- cog_basket_unresolved
- title: Session - title: Session
contents: contents:
- has_keyword("internal") - has_keyword("internal")
+35
View File
@@ -0,0 +1,35 @@
% Generated by roxygen2: do not edit by hand
% Please edit documentation in R/basket.R
\name{cog_basket_resolution}
\alias{cog_basket_resolution}
\title{Inspect basket-mode resolution sidecar}
\usage{
cog_basket_resolution(x, expand_candidates = FALSE)
}
\arguments{
\item{x}{A tibble returned by basket-mode [cog_gov_search()].}
\item{expand_candidates}{Logical. If `TRUE`, keeps the `candidates`
list-column (full-schema match candidates per input row). Default
`FALSE`.}
}
\value{
A tibble with the resolution audit trail.
}
\description{
Returns the resolution tibble attached to a basket-mode result of
[cog_gov_search()]. One row per input entry; `status` is one of
`"resolved"`, `"largest_pop"`, `"ambiguous"`, `"no_match"`. By default
the `candidates` list-column is dropped for readable printing; pass
`expand_candidates = TRUE` to keep it.
}
\examples{
\dontrun{
basket <- cog_gov_search(
name = c("Broward", "San Diego", "Notarealplace"),
state = c("FL", "CA", "NY")
)
cog_basket_resolution(basket)
cog_basket_resolution(basket, expand_candidates = TRUE)
}
}
+29
View File
@@ -0,0 +1,29 @@
% Generated by roxygen2: do not edit by hand
% Please edit documentation in R/basket.R
\name{cog_basket_unresolved}
\alias{cog_basket_unresolved}
\title{Filter a basket resolution to unresolved rows}
\usage{
cog_basket_unresolved(x)
}
\arguments{
\item{x}{A tibble returned by basket-mode [cog_gov_search()].}
}
\value{
A tibble (subset of [cog_basket_resolution()]).
}
\description{
Convenience wrapper that returns just the rows where `status` is
`"ambiguous"` or `"no_match"` — the ones the user likely wants to
refine before piping into a query verb. The `candidates` list-column
is preserved so the user can drill into ambiguous match sets.
}
\examples{
\dontrun{
basket <- cog_gov_search(
name = c("Broward", "San Diego", "Notarealplace"),
state = c("FL", "CA", "NY")
)
cog_basket_unresolved(basket)
}
}
+79 -15
View File
@@ -4,27 +4,91 @@
\alias{cog_gov_search} \alias{cog_gov_search}
\title{Search for governments by name, state, and/or type} \title{Search for governments by name, state, and/or type}
\usage{ \usage{
cog_gov_search(pattern = NULL, state = NULL, type = NULL) cog_gov_search(name = NULL, state = NULL, type = NULL)
} }
\arguments{ \arguments{
\item{pattern}{Character regex matched case-insensitively against \item{name}{Character vector of place name(s). Length 1 = utility mode;
`gov_name`. `NULL` (default) means no name filter.} length >1 = basket mode.}
\item{state}{Either a 2-letter USPS abbreviation (e.g. `"FL"`), a FIPS \item{state}{2-letter USPS abbreviation (e.g. `"FL"`), FIPS integer
integer (e.g. `12`), or `NULL`.} (e.g. `12`), or `NULL`. In basket mode, length 1 recycles across
all entries; otherwise must match `length(name)`.}
\item{type}{Government type: an integer in `0:3` or one of `"state"`, \item{type}{Government type: integer in `0:3` or one of `"state"`,
`"county"`, `"city"`, `"township"`. Passing `4`, `5`, `"county"`, `"city"`, `"township"`, or `NA`/`NULL`. Per-row optional
`"special_district"`, or `"school_district"` emits an explanatory in basket mode (recycles from length 1). Excluded types `4`/`5` (or
message and returns an empty tibble (v0.1 corpus excludes those types).} `"special_district"` / `"school_district"`) trigger an explanatory
message and an empty result.}
} }
\value{ \value{
Tibble from `canonical_fips_xwalk` sorted by `population_acs` A tibble of `canonical_fips_xwalk` rows. In utility mode, all
descending (`NULL`s last). matches sorted by `population_acs` desc. In basket mode, resolved
rows in input order, with `attr(., "resolution")` set to the
sidecar tibble.
} }
\description{ \description{
Returns rows from `canonical_fips_xwalk` matching the supplied filters. Resolves human-readable place names into rows of `canonical_fips_xwalk`,
Intended as the entry point users call to resolve a human-readable place the cross-vintage canonical-government registry. Operates in two modes:
name into one or more `canonical_govid` values before calling }
[cog_spending()] / [cog_revenue()] / etc. \details{
* **Utility mode** (single `name`, the original behavior): returns all
rows whose `gov_name` matches the regex case-insensitively, sorted by
`population_acs` descending. Useful for exploratory lookups.
* **Basket mode** (`length(name) > 1`): resolves each input row to a
single canonical govid and returns a tibble in input order, suitable
for piping straight into [cog_spending()] / [cog_revenue()] /
[cog_geographic_rollup()]. Carries an audit sidecar accessible via
[cog_basket_resolution()] / [cog_basket_unresolved()].
**Basket-mode resolution algorithm** (per input row):
1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`.
2. **Exact pass:** case-insensitive equality against `gov_name`.
Single hit -> resolved. Multiple -> step 4.
3. **Substring fallback:** case-insensitive regex against `gov_name`.
Single hit -> resolved (`match_method = "substring"`). Zero hits ->
`status = "no_match"`. Multiple hits -> step 4.
4. **Disambiguation:** if matches share one `govs_type`, pick the
largest-population row (`status = "largest_pop"`). If they span >=2
types, no row is added (`status = "ambiguous"`); the user should
re-run with `type` specified.
Resolved rows form the returned tibble in input order. Unresolved
inputs (`ambiguous` / `no_match`) appear only in the sidecar.
}
\examples{
\dontrun{
# Utility mode — exploratory regex lookup
cog_gov_search("broward", state = "FL")
# Basket mode — resolve a known cohort
basket <- cog_gov_search(
name = c("BROWARD COUNTY", "SAN DIEGO CITY", "AUSTIN CITY"),
state = c("FL", "CA", "TX")
)
basket
# Inspect resolution audit
cog_basket_resolution(basket)
# Pipe into a spending query
library(dplyr)
basket |> cog_spending(years = 2019:2020, category = "Police")
# Iteratively refine ambiguous matches
partial <- cog_gov_search(
name = c("Broward", "San Diego"), # San Diego is ambiguous
state = c("FL", "CA")
)
cog_basket_unresolved(partial)
refined <- cog_gov_search(
name = c("Broward", "San Diego"),
state = c("FL", "CA"),
type = c(NA, "city") # disambiguate
)
}
}
\seealso{
[cog_basket_resolution()], [cog_basket_unresolved()],
[cog_spending()], [cog_revenue()].
} }
+51
View File
@@ -0,0 +1,51 @@
test_that("cog_basket_resolution returns the sidecar tibble", {
basket <- suppressMessages(cog_gov_search(
name = c("Broward", "Notarealplace"),
state = c("FL", "NY")
))
res <- cog_basket_resolution(basket)
expect_s3_class(res, "tbl_df")
expect_equal(nrow(res), 2L)
expect_setequal(colnames(res), c(
"query_name", "query_state", "query_type", "status",
"match_method", "canonical_govid", "gov_name", "n_candidates"
))
})
test_that("cog_basket_resolution(expand_candidates = TRUE) keeps candidates list-col", {
basket <- suppressMessages(cog_gov_search(
name = c("Broward", "Notarealplace"),
state = c("FL", "NY")
))
res <- cog_basket_resolution(basket, expand_candidates = TRUE)
expect_true("candidates" %in% colnames(res))
expect_true(is.list(res$candidates))
})
test_that("cog_basket_resolution errors on a non-basket tibble", {
utility <- cog_gov_search("BROWARD")
expect_error(
cog_basket_resolution(utility),
regexp = "no resolution attribute"
)
})
test_that("cog_basket_unresolved filters to ambiguous and no_match", {
basket <- suppressMessages(cog_gov_search(
name = c("Broward", "San Diego", "Notarealplace"),
state = c("FL", "CA", "NY")
))
unres <- cog_basket_unresolved(basket)
expect_equal(nrow(unres), 2L)
expect_setequal(unres$status, c("ambiguous", "no_match"))
expect_true("candidates" %in% colnames(unres))
})
test_that("cog_basket_unresolved returns 0 rows when basket is clean", {
basket <- cog_gov_search(
name = c("BROWARD COUNTY", "SAN DIEGO CITY"),
state = c("FL", "CA")
)
unres <- cog_basket_unresolved(basket)
expect_equal(nrow(unres), 0L)
})
+363
View File
@@ -72,3 +72,366 @@ test_that("cog_gov_search with no filters returns full registry", {
r <- cog_gov_search() r <- cog_gov_search()
expect_gt(nrow(r), 1000L) expect_gt(nrow(r), 1000L)
}) })
# ---- basket mode internal helpers ----
test_that(".validate_basket_args recycles state from length 1", {
out <- uscogdata:::.validate_basket_args(
name = c("Broward", "San Diego", "Austin"),
state = "FL",
type = NULL
)
expect_equal(out$name, c("Broward", "San Diego", "Austin"))
expect_equal(out$state, c("FL", "FL", "FL"))
expect_equal(out$type, c(NA_character_, NA_character_, NA_character_))
})
test_that(".validate_basket_args recycles type from length 1", {
out <- uscogdata:::.validate_basket_args(
name = c("San Diego", "Oakland"),
state = "CA",
type = "city"
)
expect_equal(out$type, c("city", "city"))
})
test_that(".validate_basket_args accepts per-row state and type", {
out <- uscogdata:::.validate_basket_args(
name = c("Broward", "San Diego"),
state = c("FL", "CA"),
type = c(NA, "city")
)
expect_equal(out$state, c("FL", "CA"))
expect_equal(out$type, c(NA_character_, "city"))
})
test_that(".validate_basket_args rejects length-mismatched state", {
expect_error(
uscogdata:::.validate_basket_args(
name = c("Broward", "San Diego", "Austin"),
state = c("FL", "CA"),
type = NULL
),
regexp = "must be length 1 or 3"
)
})
test_that(".validate_basket_args rejects length-mismatched type", {
expect_error(
uscogdata:::.validate_basket_args(
name = c("Broward", "San Diego"),
state = "FL",
type = c("county", "city", "city")
),
regexp = "must be length 1 or 2"
)
})
test_that(".validate_basket_args allows NULL state and type", {
out <- uscogdata:::.validate_basket_args(
name = c("Broward", "San Diego"),
state = NULL,
type = NULL
)
expect_equal(out$state, c(NA_character_, NA_character_))
expect_equal(out$type, c(NA_character_, NA_character_))
})
test_that(".resolve_basket_row exact match returns one row", {
con <- uscogdata:::.ensure_session()
out <- uscogdata:::.resolve_basket_row(
name = "BROWARD COUNTY", state = "FL", type = NA_character_, con = con
)
expect_equal(out$status, "resolved")
expect_equal(out$match_method, "exact")
expect_equal(out$n_candidates, 1L)
expect_equal(nrow(out$row), 1L)
expect_equal(out$row$canonical_govid, "101006006")
expect_equal(out$row$gov_name, "BROWARD COUNTY")
})
test_that(".resolve_basket_row exact match is case-insensitive", {
con <- uscogdata:::.ensure_session()
out <- uscogdata:::.resolve_basket_row(
name = "broward county", state = "FL", type = NA_character_, con = con
)
expect_equal(out$status, "resolved")
expect_equal(out$match_method, "exact")
expect_equal(out$row$canonical_govid, "101006006")
})
test_that(".resolve_basket_row exact match honors per-row type", {
con <- uscogdata:::.ensure_session()
out <- uscogdata:::.resolve_basket_row(
name = "SAN DIEGO CITY", state = "CA", type = "city", con = con
)
expect_equal(out$status, "resolved")
expect_equal(out$row$canonical_govid, "052037010")
})
test_that(".resolve_basket_row substring fallback resolves single match", {
con <- uscogdata:::.ensure_session()
out <- uscogdata:::.resolve_basket_row(
name = "Broward", state = "FL", type = NA_character_, con = con
)
expect_equal(out$status, "resolved")
expect_equal(out$match_method, "substring")
expect_equal(out$n_candidates, 1L)
expect_equal(out$row$canonical_govid, "101006006")
})
test_that(".resolve_basket_row no_match returns 0-row tibble", {
con <- uscogdata:::.ensure_session()
out <- uscogdata:::.resolve_basket_row(
name = "Notarealplace", state = "NY", type = NA_character_, con = con
)
expect_equal(out$status, "no_match")
expect_true(is.na(out$match_method))
expect_equal(out$n_candidates, 0L)
expect_equal(nrow(out$row), 0L)
expect_equal(nrow(out$candidates), 0L)
})
test_that(".resolve_basket_row treats empty/whitespace name as no_match", {
con <- uscogdata:::.ensure_session()
out_empty <- uscogdata:::.resolve_basket_row(
name = "", state = "FL", type = NA_character_, con = con
)
expect_equal(out_empty$status, "no_match")
out_ws <- uscogdata:::.resolve_basket_row(
name = " ", state = "FL", type = NA_character_, con = con
)
expect_equal(out_ws$status, "no_match")
})
test_that(".resolve_basket_row largest_pop within single type", {
# FL Miami substring matches 10 cities (all govs_type = 2), largest pop
# is MIAMI CITY at 443665.
con <- uscogdata:::.ensure_session()
out <- uscogdata:::.resolve_basket_row(
name = "Miami", state = "FL", type = NA_character_, con = con
)
expect_equal(out$status, "largest_pop")
expect_equal(out$match_method, "substring")
expect_gte(out$n_candidates, 2L)
expect_equal(out$row$canonical_govid, "102013013")
expect_equal(out$row$gov_name, "MIAMI CITY")
})
test_that(".resolve_basket_row ambiguous across types", {
# SAN DIEGO substring matches both SAN DIEGO COUNTY (type 1) and
# SAN DIEGO CITY (type 2) in CA.
con <- uscogdata:::.ensure_session()
out <- uscogdata:::.resolve_basket_row(
name = "San Diego", state = "CA", type = NA_character_, con = con
)
expect_equal(out$status, "ambiguous")
expect_true(is.na(out$match_method))
expect_equal(out$n_candidates, 2L)
expect_equal(nrow(out$row), 0L)
expect_equal(nrow(out$candidates), 2L)
expect_setequal(out$candidates$govs_type, c(1L, 2L))
})
test_that(".resolve_basket_row resolves with type override on ambiguous case", {
con <- uscogdata:::.ensure_session()
out <- uscogdata:::.resolve_basket_row(
name = "San Diego", state = "CA", type = "city", con = con
)
expect_equal(out$status, "resolved")
expect_equal(out$match_method, "substring")
expect_equal(out$row$canonical_govid, "052037010")
})
# ---- basket mode public surface ----
test_that("cog_gov_search basket mode resolves clean inputs in input order", {
skip_if_no_corpus()
basket <- cog_gov_search(
name = c("BROWARD COUNTY", "SAN DIEGO CITY", "AUSTIN CITY"),
state = c("FL", "CA", "TX")
)
expect_s3_class(basket, "tbl_df")
expect_equal(nrow(basket), 3L)
expect_equal(basket$canonical_govid, c("101006006", "052037010", "442227001"))
expect_equal(basket$gov_name, c("BROWARD COUNTY", "SAN DIEGO CITY", "AUSTIN CITY"))
})
test_that("cog_gov_search basket mode attaches a resolution sidecar", {
skip_if_no_corpus()
basket <- cog_gov_search(
name = c("Broward", "San Diego"),
state = c("FL", "CA"),
type = c(NA, "city")
)
res <- attr(basket, "resolution")
expect_s3_class(res, "tbl_df")
expect_equal(nrow(res), 2L)
expect_equal(res$query_name, c("Broward", "San Diego"))
expect_equal(res$query_state, c("FL", "CA"))
expect_equal(res$query_type, c(NA_character_, "city"))
expect_equal(res$status, c("resolved", "resolved"))
expect_equal(res$match_method, c("substring", "substring"))
expect_true(is.list(res$candidates))
})
test_that("cog_gov_search basket mode skips ambiguous and no_match rows", {
skip_if_no_corpus()
basket <- suppressMessages(cog_gov_search(
name = c("Broward", "San Diego", "Notarealplace"),
state = c("FL", "CA", "NY")
))
# Broward resolves; San Diego ambiguous; Notarealplace no_match.
expect_equal(nrow(basket), 1L)
expect_equal(basket$canonical_govid, "101006006")
res <- attr(basket, "resolution")
expect_equal(nrow(res), 3L)
expect_equal(res$status, c("resolved", "ambiguous", "no_match"))
})
test_that("cog_gov_search basket mode preserves input order", {
skip_if_no_corpus()
basket <- cog_gov_search(
name = c("AUSTIN CITY", "BROWARD COUNTY", "SAN DIEGO CITY"),
state = c("TX", "FL", "CA")
)
expect_equal(basket$gov_name, c("AUSTIN CITY", "BROWARD COUNTY", "SAN DIEGO CITY"))
})
test_that("cog_gov_search basket mode recycles single state", {
skip_if_no_corpus()
basket <- cog_gov_search(
name = c("SAN DIEGO CITY", "OAKLAND CITY"),
state = "CA"
)
expect_equal(nrow(basket), 2L)
expect_equal(basket$canonical_govid, c("052037010", "052001009"))
})
test_that("cog_gov_search basket mode within-type largest_pop records candidates", {
skip_if_no_corpus()
basket <- suppressMessages(cog_gov_search(
name = c("Miami", "OAKLAND CITY"),
state = c("FL", "CA")
))
expect_equal(nrow(basket), 2L)
res <- attr(basket, "resolution")
miami_row <- res[res$query_name == "Miami", ]
expect_equal(miami_row$status, "largest_pop")
expect_equal(miami_row$canonical_govid, "102013013")
expect_gte(miami_row$n_candidates, 2L)
expect_gte(nrow(miami_row$candidates[[1]]), 2L)
})
test_that("cog_gov_search utility mode (length-1 name) has no sidecar", {
skip_if_no_corpus()
r <- cog_gov_search("BROWARD")
expect_null(attr(r, "resolution"))
expect_gte(nrow(r), 1L)
})
test_that("cog_gov_search basket mode validates argument lengths", {
expect_error(
cog_gov_search(
name = c("Broward", "San Diego", "Austin"),
state = c("FL", "CA")
),
regexp = "must be length 1 or 3"
)
})
# ---- basket mode summary message ----
test_that("cog_gov_search basket mode is silent on clean basket", {
skip_if_no_corpus()
expect_message(
cog_gov_search(
name = c("BROWARD COUNTY", "SAN DIEGO CITY"),
state = c("FL", "CA")
),
regexp = NA # NA = expect no message
)
})
test_that("cog_gov_search basket mode reports breakdown on partial basket", {
skip_if_no_corpus()
expect_message(
cog_gov_search(
name = c("Broward", "San Diego", "Notarealplace"),
state = c("FL", "CA", "NY")
),
regexp = "Basket resolved 1 of 3"
)
})
test_that("cog_gov_search basket mode message points to the sidecar accessor", {
skip_if_no_corpus()
expect_message(
cog_gov_search(
name = c("Broward", "Notarealplace"),
state = c("FL", "NY")
),
regexp = "cog_basket_resolution"
)
})
# ---- F1: per-row excluded type soft-fail ----
test_that(".resolve_basket_row treats excluded type as no_match (not abort)", {
con <- uscogdata:::.ensure_session()
out <- uscogdata:::.resolve_basket_row(
name = "Some District", state = "CA", type = "special_district", con = con
)
expect_equal(out$status, "no_match")
expect_true(is.na(out$match_method))
expect_equal(out$n_candidates, 0L)
})
test_that("cog_gov_search basket mode skips per-row excluded type without aborting", {
basket <- suppressMessages(cog_gov_search(
name = c("BROWARD COUNTY", "Some District"),
state = c("FL", "FL"),
type = c(NA, "special_district")
))
# Broward should resolve; the special_district row should be no_match.
expect_equal(nrow(basket), 1L)
expect_equal(basket$canonical_govid, "101006006")
res <- attr(basket, "resolution")
expect_equal(res$status, c("resolved", "no_match"))
# query_type should record what the user passed for the excluded-type row
expect_equal(res$query_type, c(NA_character_, "special_district"))
})
# ---- F2: malformed regex name soft-fail ----
test_that(".resolve_basket_row treats malformed regex name as no_match", {
con <- uscogdata:::.ensure_session()
# Unbalanced parens would be a regex parse error if not escaped.
out <- uscogdata:::.resolve_basket_row(
name = "San(Diego", state = "CA", type = NA_character_, con = con
)
expect_equal(out$status, "no_match")
})
test_that(".resolve_basket_row escapes regex metacharacters in name", {
con <- uscogdata:::.ensure_session()
# Confirm that names with various metacharacters don't error.
expect_no_error(uscogdata:::.resolve_basket_row(
name = "Foo*Bar+Baz", state = "FL", type = NA_character_, con = con
))
})
# ---- F3: all-no-match basket public surface ----
test_that("cog_gov_search basket all-no-match returns 0-row tibble with full sidecar", {
basket <- suppressMessages(cog_gov_search(
name = c("Notarealplace1", "Notarealplace2"),
state = c("NY", "CA")
))
expect_equal(nrow(basket), 0L)
expect_true("canonical_govid" %in% names(basket))
res <- attr(basket, "resolution")
expect_equal(nrow(res), 2L)
expect_true(all(res$status == "no_match"))
})
+12
View File
@@ -116,3 +116,15 @@ test_that("cog_spending rejects data.frame without canonical_govid column", {
bad <- tibble::tibble(foo = "bar") bad <- tibble::tibble(foo = "bar")
expect_error(cog_spending(bad, 2020L), "canonical_govid") expect_error(cog_spending(bad, 2020L), "canonical_govid")
}) })
test_that("cog_spending accepts a basket-mode cog_gov_search result", {
skip_if_no_corpus()
basket <- cog_gov_search(
name = c("BROWARD COUNTY", "SAN DIEGO COUNTY"),
state = c("FL", "CA")
)
expect_equal(nrow(basket), 2L)
spending <- cog_spending(basket, years = 2019:2020, category = "Police")
expect_s3_class(spending, "tbl_df")
expect_setequal(unique(spending$canonical_govid), basket$canonical_govid)
})