Merge pull request 'feat(search): basket mode for cog_gov_search()' (#1) from feat/cog-gov-search-basket-mode into main
R-CMD-check / check (push) Successful in 1m34s
R-CMD-check / check (push) Successful in 1m34s
This commit was merged in pull request #1.
This commit is contained in:
@@ -1,5 +1,7 @@
|
|||||||
# Generated by roxygen2: do not edit by hand
|
# Generated by roxygen2: do not edit by hand
|
||||||
|
|
||||||
|
export(cog_basket_resolution)
|
||||||
|
export(cog_basket_unresolved)
|
||||||
export(cog_categories)
|
export(cog_categories)
|
||||||
export(cog_explain)
|
export(cog_explain)
|
||||||
export(cog_find_peers)
|
export(cog_find_peers)
|
||||||
|
|||||||
@@ -0,0 +1,21 @@
|
|||||||
|
# uscogdata 0.1.0 (development)
|
||||||
|
|
||||||
|
## New features
|
||||||
|
|
||||||
|
* `cog_gov_search()` gains a **basket mode**: passing vector `name`
|
||||||
|
/ `state` / `type` arguments resolves multiple place names in one
|
||||||
|
call and returns a tibble of canonical rows in input order, ready
|
||||||
|
to pipe into `cog_spending()` / `cog_revenue()`. Per-row resolution
|
||||||
|
follows an exact-then-substring matching algorithm with deterministic
|
||||||
|
disambiguation; ambiguous and missing entries are surfaced via a
|
||||||
|
sidecar audit tibble plus a single console summary message.
|
||||||
|
* New exports `cog_basket_resolution()` and `cog_basket_unresolved()`
|
||||||
|
expose the basket sidecar for iterative query refinement.
|
||||||
|
|
||||||
|
## Breaking changes
|
||||||
|
|
||||||
|
* The first formal of `cog_gov_search()` was renamed from `pattern`
|
||||||
|
to `name`. All existing call sites in `cog_explorer/` and the
|
||||||
|
package itself use positional first-arg, so this rename is
|
||||||
|
non-breaking in practice. Callers that pass `pattern = ...` by name
|
||||||
|
must update to `name = ...`.
|
||||||
+138
@@ -0,0 +1,138 @@
|
|||||||
|
# R/basket.R
|
||||||
|
#
|
||||||
|
# Internals supporting the basket-mode sidecar (constructed in
|
||||||
|
# .resolve_basket() — see R/search.R) plus the user-facing accessors
|
||||||
|
# cog_basket_resolution() and cog_basket_unresolved() (added in a
|
||||||
|
# later step).
|
||||||
|
|
||||||
|
# Build the sidecar tibble. One row per input; carries query_*, status,
|
||||||
|
# match_method, canonical_govid, gov_name, n_candidates, and a list-col
|
||||||
|
# `candidates` of full-schema match-candidate tibbles.
|
||||||
|
#' @noRd
|
||||||
|
.build_sidecar <- function(args, resolved) {
|
||||||
|
type_label <- unname(vapply(args$type, function(t) {
|
||||||
|
if (is.na(t)) NA_character_ else .type_to_label(t)
|
||||||
|
}, character(1)))
|
||||||
|
|
||||||
|
status <- vapply(resolved, `[[`, character(1), "status")
|
||||||
|
method <- vapply(resolved, `[[`, character(1), "match_method")
|
||||||
|
ncand <- vapply(resolved, `[[`, integer(1), "n_candidates")
|
||||||
|
govid <- vapply(resolved, function(r) {
|
||||||
|
if (nrow(r$row) == 0L) NA_character_ else r$row$canonical_govid[1L]
|
||||||
|
}, character(1))
|
||||||
|
gname <- vapply(resolved, function(r) {
|
||||||
|
if (nrow(r$row) == 0L) NA_character_ else r$row$gov_name[1L]
|
||||||
|
}, character(1))
|
||||||
|
cands <- lapply(resolved, `[[`, "candidates")
|
||||||
|
|
||||||
|
tibble::tibble(
|
||||||
|
query_name = args$name,
|
||||||
|
query_state = args$state,
|
||||||
|
query_type = type_label,
|
||||||
|
status = status,
|
||||||
|
match_method = method,
|
||||||
|
canonical_govid = govid,
|
||||||
|
gov_name = gname,
|
||||||
|
n_candidates = ncand,
|
||||||
|
candidates = cands
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Convert a type input (integer-like or label) into the canonical label
|
||||||
|
# string used in the sidecar query_type column. Excluded types (4/5 /
|
||||||
|
# special_district / school_district) are returned as-is so the sidecar
|
||||||
|
# records what the user passed without calling .coerce_type() (which aborts).
|
||||||
|
#' @noRd
|
||||||
|
.type_to_label <- function(type) {
|
||||||
|
if (.is_excluded_type(type)) return(as.character(type))
|
||||||
|
int_type <- .coerce_type(type)
|
||||||
|
unname(c("0" = "state", "1" = "county", "2" = "city", "3" = "township")[[as.character(int_type)]])
|
||||||
|
}
|
||||||
|
|
||||||
|
# Single post-resolution summary message. Silent on clean baskets;
|
||||||
|
# emits one cli_inform with two-line body otherwise.
|
||||||
|
#' @noRd
|
||||||
|
.basket_summary_message <- function(sidecar) {
|
||||||
|
status <- sidecar$status
|
||||||
|
n_input <- length(status)
|
||||||
|
n_basket <- sum(status %in% c("resolved", "largest_pop"))
|
||||||
|
n_amb <- sum(status == "ambiguous")
|
||||||
|
n_nm <- sum(status == "no_match")
|
||||||
|
n_lp <- sum(status == "largest_pop")
|
||||||
|
|
||||||
|
if (n_amb == 0L && n_nm == 0L && n_lp == 0L) return(invisible(NULL))
|
||||||
|
|
||||||
|
parts <- c(
|
||||||
|
if (n_amb > 0L) sprintf("%d ambiguous", n_amb),
|
||||||
|
if (n_nm > 0L) sprintf("%d with no match", n_nm),
|
||||||
|
if (n_lp > 0L) sprintf("%d used largest-population fallback", n_lp)
|
||||||
|
)
|
||||||
|
|
||||||
|
cli::cli_inform(c(
|
||||||
|
i = sprintf("Basket resolved %d of %d entries.", n_basket, n_input),
|
||||||
|
i = paste(parts, collapse = ", "),
|
||||||
|
i = "Inspect with `cog_basket_resolution(result)` or filter to problem rows with `cog_basket_unresolved(result)`."
|
||||||
|
))
|
||||||
|
invisible(NULL)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Inspect basket-mode resolution sidecar
|
||||||
|
#'
|
||||||
|
#' Returns the resolution tibble attached to a basket-mode result of
|
||||||
|
#' [cog_gov_search()]. One row per input entry; `status` is one of
|
||||||
|
#' `"resolved"`, `"largest_pop"`, `"ambiguous"`, `"no_match"`. By default
|
||||||
|
#' the `candidates` list-column is dropped for readable printing; pass
|
||||||
|
#' `expand_candidates = TRUE` to keep it.
|
||||||
|
#'
|
||||||
|
#' @param x A tibble returned by basket-mode [cog_gov_search()].
|
||||||
|
#' @param expand_candidates Logical. If `TRUE`, keeps the `candidates`
|
||||||
|
#' list-column (full-schema match candidates per input row). Default
|
||||||
|
#' `FALSE`.
|
||||||
|
#' @return A tibble with the resolution audit trail.
|
||||||
|
#' @examples
|
||||||
|
#' \dontrun{
|
||||||
|
#' basket <- cog_gov_search(
|
||||||
|
#' name = c("Broward", "San Diego", "Notarealplace"),
|
||||||
|
#' state = c("FL", "CA", "NY")
|
||||||
|
#' )
|
||||||
|
#' cog_basket_resolution(basket)
|
||||||
|
#' cog_basket_resolution(basket, expand_candidates = TRUE)
|
||||||
|
#' }
|
||||||
|
#' @export
|
||||||
|
cog_basket_resolution <- function(x, expand_candidates = FALSE) {
|
||||||
|
res <- attr(x, "resolution")
|
||||||
|
if (is.null(res)) {
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"`x` has no resolution attribute.",
|
||||||
|
i = "Pass the result of basket-mode `cog_gov_search()` (length(name) > 1).",
|
||||||
|
i = "Single-name (utility) results do not carry a sidecar."
|
||||||
|
))
|
||||||
|
}
|
||||||
|
if (!isTRUE(expand_candidates)) {
|
||||||
|
res$candidates <- NULL
|
||||||
|
}
|
||||||
|
res
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Filter a basket resolution to unresolved rows
|
||||||
|
#'
|
||||||
|
#' Convenience wrapper that returns just the rows where `status` is
|
||||||
|
#' `"ambiguous"` or `"no_match"` — the ones the user likely wants to
|
||||||
|
#' refine before piping into a query verb. The `candidates` list-column
|
||||||
|
#' is preserved so the user can drill into ambiguous match sets.
|
||||||
|
#'
|
||||||
|
#' @param x A tibble returned by basket-mode [cog_gov_search()].
|
||||||
|
#' @return A tibble (subset of [cog_basket_resolution()]).
|
||||||
|
#' @examples
|
||||||
|
#' \dontrun{
|
||||||
|
#' basket <- cog_gov_search(
|
||||||
|
#' name = c("Broward", "San Diego", "Notarealplace"),
|
||||||
|
#' state = c("FL", "CA", "NY")
|
||||||
|
#' )
|
||||||
|
#' cog_basket_unresolved(basket)
|
||||||
|
#' }
|
||||||
|
#' @export
|
||||||
|
cog_basket_unresolved <- function(x) {
|
||||||
|
res <- cog_basket_resolution(x, expand_candidates = TRUE)
|
||||||
|
res[res$status %in% c("ambiguous", "no_match"), , drop = FALSE]
|
||||||
|
}
|
||||||
+279
-20
@@ -2,40 +2,105 @@
|
|||||||
|
|
||||||
#' Search for governments by name, state, and/or type
|
#' Search for governments by name, state, and/or type
|
||||||
#'
|
#'
|
||||||
#' Returns rows from `canonical_fips_xwalk` matching the supplied filters.
|
#' Resolves human-readable place names into rows of `canonical_fips_xwalk`,
|
||||||
#' Intended as the entry point users call to resolve a human-readable place
|
#' the cross-vintage canonical-government registry. Operates in two modes:
|
||||||
#' name into one or more `canonical_govid` values before calling
|
|
||||||
#' [cog_spending()] / [cog_revenue()] / etc.
|
|
||||||
#'
|
#'
|
||||||
#' @param pattern Character regex matched case-insensitively against
|
#' * **Utility mode** (single `name`, the original behavior): returns all
|
||||||
#' `gov_name`. `NULL` (default) means no name filter.
|
#' rows whose `gov_name` matches the regex case-insensitively, sorted by
|
||||||
#' @param state Either a 2-letter USPS abbreviation (e.g. `"FL"`), a FIPS
|
#' `population_acs` descending. Useful for exploratory lookups.
|
||||||
#' integer (e.g. `12`), or `NULL`.
|
#' * **Basket mode** (`length(name) > 1`): resolves each input row to a
|
||||||
#' @param type Government type: an integer in `0:3` or one of `"state"`,
|
#' single canonical govid and returns a tibble in input order, suitable
|
||||||
#' `"county"`, `"city"`, `"township"`. Passing `4`, `5`,
|
#' for piping straight into [cog_spending()] / [cog_revenue()] /
|
||||||
#' `"special_district"`, or `"school_district"` emits an explanatory
|
#' [cog_geographic_rollup()]. Carries an audit sidecar accessible via
|
||||||
#' message and returns an empty tibble (v0.1 corpus excludes those types).
|
#' [cog_basket_resolution()] / [cog_basket_unresolved()].
|
||||||
#' @return Tibble from `canonical_fips_xwalk` sorted by `population_acs`
|
#'
|
||||||
#' descending (`NULL`s last).
|
#' @details
|
||||||
|
#' **Basket-mode resolution algorithm** (per input row):
|
||||||
|
#' 1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`.
|
||||||
|
#' 2. **Exact pass:** case-insensitive equality against `gov_name`.
|
||||||
|
#' Single hit -> resolved. Multiple -> step 4.
|
||||||
|
#' 3. **Substring fallback:** case-insensitive regex against `gov_name`.
|
||||||
|
#' Single hit -> resolved (`match_method = "substring"`). Zero hits ->
|
||||||
|
#' `status = "no_match"`. Multiple hits -> step 4.
|
||||||
|
#' 4. **Disambiguation:** if matches share one `govs_type`, pick the
|
||||||
|
#' largest-population row (`status = "largest_pop"`). If they span >=2
|
||||||
|
#' types, no row is added (`status = "ambiguous"`); the user should
|
||||||
|
#' re-run with `type` specified.
|
||||||
|
#'
|
||||||
|
#' Resolved rows form the returned tibble in input order. Unresolved
|
||||||
|
#' inputs (`ambiguous` / `no_match`) appear only in the sidecar.
|
||||||
|
#'
|
||||||
|
#' @param name Character vector of place name(s). Length 1 = utility mode;
|
||||||
|
#' length >1 = basket mode.
|
||||||
|
#' @param state 2-letter USPS abbreviation (e.g. `"FL"`), FIPS integer
|
||||||
|
#' (e.g. `12`), or `NULL`. In basket mode, length 1 recycles across
|
||||||
|
#' all entries; otherwise must match `length(name)`.
|
||||||
|
#' @param type Government type: integer in `0:3` or one of `"state"`,
|
||||||
|
#' `"county"`, `"city"`, `"township"`, or `NA`/`NULL`. Per-row optional
|
||||||
|
#' in basket mode (recycles from length 1). Excluded types `4`/`5` (or
|
||||||
|
#' `"special_district"` / `"school_district"`) trigger an explanatory
|
||||||
|
#' message and an empty result.
|
||||||
|
#' @return A tibble of `canonical_fips_xwalk` rows. In utility mode, all
|
||||||
|
#' matches sorted by `population_acs` desc. In basket mode, resolved
|
||||||
|
#' rows in input order, with `attr(., "resolution")` set to the
|
||||||
|
#' sidecar tibble.
|
||||||
|
#' @seealso [cog_basket_resolution()], [cog_basket_unresolved()],
|
||||||
|
#' [cog_spending()], [cog_revenue()].
|
||||||
|
#' @examples
|
||||||
|
#' \dontrun{
|
||||||
|
#' # Utility mode — exploratory regex lookup
|
||||||
|
#' cog_gov_search("broward", state = "FL")
|
||||||
|
#'
|
||||||
|
#' # Basket mode — resolve a known cohort
|
||||||
|
#' basket <- cog_gov_search(
|
||||||
|
#' name = c("BROWARD COUNTY", "SAN DIEGO CITY", "AUSTIN CITY"),
|
||||||
|
#' state = c("FL", "CA", "TX")
|
||||||
|
#' )
|
||||||
|
#' basket
|
||||||
|
#'
|
||||||
|
#' # Inspect resolution audit
|
||||||
|
#' cog_basket_resolution(basket)
|
||||||
|
#'
|
||||||
|
#' # Pipe into a spending query
|
||||||
|
#' library(dplyr)
|
||||||
|
#' basket |> cog_spending(years = 2019:2020, category = "Police")
|
||||||
|
#'
|
||||||
|
#' # Iteratively refine ambiguous matches
|
||||||
|
#' partial <- cog_gov_search(
|
||||||
|
#' name = c("Broward", "San Diego"), # San Diego is ambiguous
|
||||||
|
#' state = c("FL", "CA")
|
||||||
|
#' )
|
||||||
|
#' cog_basket_unresolved(partial)
|
||||||
|
#' refined <- cog_gov_search(
|
||||||
|
#' name = c("Broward", "San Diego"),
|
||||||
|
#' state = c("FL", "CA"),
|
||||||
|
#' type = c(NA, "city") # disambiguate
|
||||||
|
#' )
|
||||||
|
#' }
|
||||||
#' @export
|
#' @export
|
||||||
cog_gov_search <- function(pattern = NULL, state = NULL, type = NULL) {
|
cog_gov_search <- function(name = NULL, state = NULL, type = NULL) {
|
||||||
if (!is.null(type) && .is_excluded_type(type)) {
|
if (!is.null(type) && length(type) == 1L && .is_excluded_type(type)) {
|
||||||
cli::cli_inform(c(
|
cli::cli_inform(c(
|
||||||
i = "v0.1 covers gov_types 0-3 (state/county/city/township) only.",
|
i = "v0.1 covers gov_types 0-3 (state/county/city/township) only.",
|
||||||
i = "Types 4 (special districts) and 5 (school districts) are excluded; see vignette('coverage-scope')."
|
i = "Types 4 (special districts) and 5 (school districts) are excluded; see vignette('coverage-scope')."
|
||||||
))
|
))
|
||||||
return(.empty_xwalk_tibble())
|
return(.empty_xwalk_tibble())
|
||||||
}
|
}
|
||||||
|
|
||||||
con <- .ensure_session()
|
con <- .ensure_session()
|
||||||
|
|
||||||
|
if (length(name) > 1L) {
|
||||||
|
return(.resolve_basket(name = name, state = state, type = type, con = con))
|
||||||
|
}
|
||||||
|
|
||||||
preds <- character(0)
|
preds <- character(0)
|
||||||
if (!is.null(pattern)) {
|
if (!is.null(name)) {
|
||||||
if (!is.character(pattern) || length(pattern) != 1L) {
|
if (!is.character(name) || length(name) != 1L) {
|
||||||
cli::cli_abort("`pattern` must be a length-1 character string.")
|
cli::cli_abort("`name` must be a length-1 character string.")
|
||||||
}
|
}
|
||||||
preds <- c(preds,
|
preds <- c(preds,
|
||||||
sprintf("regexp_matches(gov_name, %s, 'i')",
|
sprintf("regexp_matches(gov_name, %s, 'i')",
|
||||||
.sql_lit_chr(pattern)))
|
.sql_lit_chr(name)))
|
||||||
}
|
}
|
||||||
if (!is.null(state)) {
|
if (!is.null(state)) {
|
||||||
st_fips <- .coerce_state_to_fips(state)
|
st_fips <- .coerce_state_to_fips(state)
|
||||||
@@ -67,6 +132,14 @@ cog_gov_search <- function(pattern = NULL, state = NULL, type = NULL) {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#' @noRd
|
||||||
|
.escape_regex <- function(x) {
|
||||||
|
# Backslash-escape POSIX regex metacharacters so `name` is treated as a
|
||||||
|
# literal substring in the DuckDB regexp_matches call (substring fallback
|
||||||
|
# only; utility-mode intentionally preserves regex behavior).
|
||||||
|
gsub("([\\^$.|?*+(){}\\[\\]])", "\\\\\\1", x, perl = TRUE)
|
||||||
|
}
|
||||||
|
|
||||||
#' @noRd
|
#' @noRd
|
||||||
.is_excluded_type <- function(type) {
|
.is_excluded_type <- function(type) {
|
||||||
excluded <- c("4", "5", "special_district", "school_district")
|
excluded <- c("4", "5", "special_district", "school_district")
|
||||||
@@ -120,3 +193,189 @@ cog_gov_search <- function(pattern = NULL, state = NULL, type = NULL) {
|
|||||||
WV = "54", WI = "55", WY = "56",
|
WV = "54", WI = "55", WY = "56",
|
||||||
AS = "60", GU = "66", MP = "69", PR = "72", VI = "78"
|
AS = "60", GU = "66", MP = "69", PR = "72", VI = "78"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# Validate basket-mode inputs. Returns a list with normalized character
|
||||||
|
# vectors `name`, `state`, `type`, all of length n = length(name).
|
||||||
|
# `state` and `type` of length 1 are recycled; lengths must be 1 or n
|
||||||
|
# otherwise. NULL state/type become a vector of NA_character_.
|
||||||
|
#' @noRd
|
||||||
|
.validate_basket_args <- function(name, state, type) {
|
||||||
|
if (!is.character(name)) {
|
||||||
|
cli::cli_abort("`name` must be a character vector.")
|
||||||
|
}
|
||||||
|
n <- length(name)
|
||||||
|
|
||||||
|
state_norm <- if (is.null(state)) {
|
||||||
|
rep(NA_character_, n)
|
||||||
|
} else if (length(state) == 1L) {
|
||||||
|
rep(as.character(state), n)
|
||||||
|
} else if (length(state) == n) {
|
||||||
|
as.character(state)
|
||||||
|
} else {
|
||||||
|
cli::cli_abort(
|
||||||
|
"`state` must be length 1 or {n} (length of `name`); got {length(state)}."
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
type_norm <- if (is.null(type)) {
|
||||||
|
rep(NA_character_, n)
|
||||||
|
} else if (length(type) == 1L) {
|
||||||
|
rep(as.character(type), n)
|
||||||
|
} else if (length(type) == n) {
|
||||||
|
as.character(type)
|
||||||
|
} else {
|
||||||
|
cli::cli_abort(
|
||||||
|
"`type` must be length 1 or {n} (length of `name`); got {length(type)}."
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
list(name = name, state = state_norm, type = type_norm)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Resolve a single basket-mode input row. Returns a list with components:
|
||||||
|
# status : "resolved" | "largest_pop" | "ambiguous" | "no_match"
|
||||||
|
# match_method : "exact" | "substring" | NA_character_
|
||||||
|
# n_candidates : int
|
||||||
|
# row : tibble (single resolved row, or 0-row tibble for unresolved)
|
||||||
|
# candidates : tibble (all rows that matched, for sidecar)
|
||||||
|
# Internal use only; takes an active DuckDB connection to reuse the session.
|
||||||
|
#' @noRd
|
||||||
|
.resolve_basket_row <- function(name, state, type, con) {
|
||||||
|
# Short-circuit: empty/whitespace name -> no_match without SQL.
|
||||||
|
if (!nzchar(trimws(name))) {
|
||||||
|
empty <- .empty_xwalk_tibble()
|
||||||
|
return(list(
|
||||||
|
status = "no_match",
|
||||||
|
match_method = NA_character_,
|
||||||
|
n_candidates = 0L,
|
||||||
|
row = empty,
|
||||||
|
candidates = empty
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
|
# Short-circuit: excluded type (4/5 / special_district / school_district)
|
||||||
|
# -> no_match without SQL, preserving soft-fail contract.
|
||||||
|
if (!is.na(type) && .is_excluded_type(type)) {
|
||||||
|
empty <- .empty_xwalk_tibble()
|
||||||
|
return(list(
|
||||||
|
status = "no_match",
|
||||||
|
match_method = NA_character_,
|
||||||
|
n_candidates = 0L,
|
||||||
|
row = empty,
|
||||||
|
candidates = empty
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
|
preds <- character(0)
|
||||||
|
if (!is.na(state)) {
|
||||||
|
st_fips <- .coerce_state_to_fips(state)
|
||||||
|
preds <- c(preds, sprintf("fips_state = %s", .sql_lit_chr(st_fips)))
|
||||||
|
}
|
||||||
|
if (!is.na(type)) {
|
||||||
|
int_type <- .coerce_type(type)
|
||||||
|
preds <- c(preds, sprintf("govs_type = %d", int_type))
|
||||||
|
}
|
||||||
|
base_where <- if (length(preds) == 0L) "" else paste("WHERE", paste(preds, collapse = " AND "))
|
||||||
|
conj <- if (nzchar(base_where)) "AND" else "WHERE"
|
||||||
|
|
||||||
|
exact_sql <- paste(
|
||||||
|
"SELECT * FROM canonical_fips_xwalk",
|
||||||
|
base_where,
|
||||||
|
conj,
|
||||||
|
sprintf("LOWER(gov_name) = LOWER(%s)", .sql_lit_chr(name))
|
||||||
|
)
|
||||||
|
exact <- tibble::as_tibble(DBI::dbGetQuery(con, exact_sql))
|
||||||
|
|
||||||
|
if (nrow(exact) == 1L) {
|
||||||
|
return(list(
|
||||||
|
status = "resolved",
|
||||||
|
match_method = "exact",
|
||||||
|
n_candidates = 1L,
|
||||||
|
row = exact,
|
||||||
|
candidates = exact
|
||||||
|
))
|
||||||
|
}
|
||||||
|
if (nrow(exact) > 1L) {
|
||||||
|
return(.disambiguate(exact, method = "exact"))
|
||||||
|
}
|
||||||
|
|
||||||
|
sub_sql <- paste(
|
||||||
|
"SELECT * FROM canonical_fips_xwalk",
|
||||||
|
base_where,
|
||||||
|
conj,
|
||||||
|
sprintf("regexp_matches(gov_name, %s, 'i')", .sql_lit_chr(.escape_regex(name)))
|
||||||
|
)
|
||||||
|
sub <- tibble::as_tibble(DBI::dbGetQuery(con, sub_sql))
|
||||||
|
|
||||||
|
if (nrow(sub) == 0L) {
|
||||||
|
return(list(
|
||||||
|
status = "no_match",
|
||||||
|
match_method = NA_character_,
|
||||||
|
n_candidates = 0L,
|
||||||
|
row = sub,
|
||||||
|
candidates = sub
|
||||||
|
))
|
||||||
|
}
|
||||||
|
if (nrow(sub) == 1L) {
|
||||||
|
return(list(
|
||||||
|
status = "resolved",
|
||||||
|
match_method = "substring",
|
||||||
|
n_candidates = 1L,
|
||||||
|
row = sub,
|
||||||
|
candidates = sub
|
||||||
|
))
|
||||||
|
}
|
||||||
|
.disambiguate(sub, method = "substring")
|
||||||
|
}
|
||||||
|
|
||||||
|
# Disambiguate a multi-row match set. Either picks the largest-pop row
|
||||||
|
# (within single-type) or returns an ambiguous result with no basket row.
|
||||||
|
#' @noRd
|
||||||
|
.disambiguate <- function(matches, method) {
|
||||||
|
types <- unique(matches$govs_type)
|
||||||
|
if (length(types) == 1L) {
|
||||||
|
pick <- matches[order(-matches$population_acs, na.last = TRUE), , drop = FALSE][1L, , drop = FALSE]
|
||||||
|
return(list(
|
||||||
|
status = "largest_pop",
|
||||||
|
match_method = method,
|
||||||
|
n_candidates = nrow(matches),
|
||||||
|
row = pick,
|
||||||
|
candidates = matches
|
||||||
|
))
|
||||||
|
}
|
||||||
|
empty <- matches[0, , drop = FALSE]
|
||||||
|
list(
|
||||||
|
status = "ambiguous",
|
||||||
|
match_method = NA_character_,
|
||||||
|
n_candidates = nrow(matches),
|
||||||
|
row = empty,
|
||||||
|
candidates = matches
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Orchestrates basket-mode resolution: validate, per-row resolve,
|
||||||
|
# assemble the basket tibble + sidecar, attach the sidecar as an attr.
|
||||||
|
#' @noRd
|
||||||
|
.resolve_basket <- function(name, state, type, con) {
|
||||||
|
args <- .validate_basket_args(name = name, state = state, type = type)
|
||||||
|
n <- length(args$name)
|
||||||
|
|
||||||
|
resolved <- vector("list", n)
|
||||||
|
for (i in seq_len(n)) {
|
||||||
|
resolved[[i]] <- .resolve_basket_row(
|
||||||
|
name = args$name[i],
|
||||||
|
state = args$state[i],
|
||||||
|
type = args$type[i],
|
||||||
|
con = con
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
basket_rows <- lapply(resolved, function(r) r$row)
|
||||||
|
basket <- dplyr::bind_rows(basket_rows[vapply(basket_rows, function(r) nrow(r) > 0L, logical(1))])
|
||||||
|
if (nrow(basket) == 0L) basket <- .empty_xwalk_tibble()
|
||||||
|
|
||||||
|
sidecar <- .build_sidecar(args, resolved)
|
||||||
|
attr(basket, "resolution") <- sidecar
|
||||||
|
.basket_summary_message(sidecar)
|
||||||
|
basket
|
||||||
|
}
|
||||||
|
|||||||
@@ -3,6 +3,12 @@ template:
|
|||||||
bootstrap: 5
|
bootstrap: 5
|
||||||
|
|
||||||
reference:
|
reference:
|
||||||
|
- title: Search & basket
|
||||||
|
desc: Resolve place names into canonical govids.
|
||||||
|
contents:
|
||||||
|
- cog_gov_search
|
||||||
|
- cog_basket_resolution
|
||||||
|
- cog_basket_unresolved
|
||||||
- title: Session
|
- title: Session
|
||||||
contents:
|
contents:
|
||||||
- has_keyword("internal")
|
- has_keyword("internal")
|
||||||
|
|||||||
@@ -0,0 +1,35 @@
|
|||||||
|
% Generated by roxygen2: do not edit by hand
|
||||||
|
% Please edit documentation in R/basket.R
|
||||||
|
\name{cog_basket_resolution}
|
||||||
|
\alias{cog_basket_resolution}
|
||||||
|
\title{Inspect basket-mode resolution sidecar}
|
||||||
|
\usage{
|
||||||
|
cog_basket_resolution(x, expand_candidates = FALSE)
|
||||||
|
}
|
||||||
|
\arguments{
|
||||||
|
\item{x}{A tibble returned by basket-mode [cog_gov_search()].}
|
||||||
|
|
||||||
|
\item{expand_candidates}{Logical. If `TRUE`, keeps the `candidates`
|
||||||
|
list-column (full-schema match candidates per input row). Default
|
||||||
|
`FALSE`.}
|
||||||
|
}
|
||||||
|
\value{
|
||||||
|
A tibble with the resolution audit trail.
|
||||||
|
}
|
||||||
|
\description{
|
||||||
|
Returns the resolution tibble attached to a basket-mode result of
|
||||||
|
[cog_gov_search()]. One row per input entry; `status` is one of
|
||||||
|
`"resolved"`, `"largest_pop"`, `"ambiguous"`, `"no_match"`. By default
|
||||||
|
the `candidates` list-column is dropped for readable printing; pass
|
||||||
|
`expand_candidates = TRUE` to keep it.
|
||||||
|
}
|
||||||
|
\examples{
|
||||||
|
\dontrun{
|
||||||
|
basket <- cog_gov_search(
|
||||||
|
name = c("Broward", "San Diego", "Notarealplace"),
|
||||||
|
state = c("FL", "CA", "NY")
|
||||||
|
)
|
||||||
|
cog_basket_resolution(basket)
|
||||||
|
cog_basket_resolution(basket, expand_candidates = TRUE)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,29 @@
|
|||||||
|
% Generated by roxygen2: do not edit by hand
|
||||||
|
% Please edit documentation in R/basket.R
|
||||||
|
\name{cog_basket_unresolved}
|
||||||
|
\alias{cog_basket_unresolved}
|
||||||
|
\title{Filter a basket resolution to unresolved rows}
|
||||||
|
\usage{
|
||||||
|
cog_basket_unresolved(x)
|
||||||
|
}
|
||||||
|
\arguments{
|
||||||
|
\item{x}{A tibble returned by basket-mode [cog_gov_search()].}
|
||||||
|
}
|
||||||
|
\value{
|
||||||
|
A tibble (subset of [cog_basket_resolution()]).
|
||||||
|
}
|
||||||
|
\description{
|
||||||
|
Convenience wrapper that returns just the rows where `status` is
|
||||||
|
`"ambiguous"` or `"no_match"` — the ones the user likely wants to
|
||||||
|
refine before piping into a query verb. The `candidates` list-column
|
||||||
|
is preserved so the user can drill into ambiguous match sets.
|
||||||
|
}
|
||||||
|
\examples{
|
||||||
|
\dontrun{
|
||||||
|
basket <- cog_gov_search(
|
||||||
|
name = c("Broward", "San Diego", "Notarealplace"),
|
||||||
|
state = c("FL", "CA", "NY")
|
||||||
|
)
|
||||||
|
cog_basket_unresolved(basket)
|
||||||
|
}
|
||||||
|
}
|
||||||
+79
-15
@@ -4,27 +4,91 @@
|
|||||||
\alias{cog_gov_search}
|
\alias{cog_gov_search}
|
||||||
\title{Search for governments by name, state, and/or type}
|
\title{Search for governments by name, state, and/or type}
|
||||||
\usage{
|
\usage{
|
||||||
cog_gov_search(pattern = NULL, state = NULL, type = NULL)
|
cog_gov_search(name = NULL, state = NULL, type = NULL)
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
\item{pattern}{Character regex matched case-insensitively against
|
\item{name}{Character vector of place name(s). Length 1 = utility mode;
|
||||||
`gov_name`. `NULL` (default) means no name filter.}
|
length >1 = basket mode.}
|
||||||
|
|
||||||
\item{state}{Either a 2-letter USPS abbreviation (e.g. `"FL"`), a FIPS
|
\item{state}{2-letter USPS abbreviation (e.g. `"FL"`), FIPS integer
|
||||||
integer (e.g. `12`), or `NULL`.}
|
(e.g. `12`), or `NULL`. In basket mode, length 1 recycles across
|
||||||
|
all entries; otherwise must match `length(name)`.}
|
||||||
|
|
||||||
\item{type}{Government type: an integer in `0:3` or one of `"state"`,
|
\item{type}{Government type: integer in `0:3` or one of `"state"`,
|
||||||
`"county"`, `"city"`, `"township"`. Passing `4`, `5`,
|
`"county"`, `"city"`, `"township"`, or `NA`/`NULL`. Per-row optional
|
||||||
`"special_district"`, or `"school_district"` emits an explanatory
|
in basket mode (recycles from length 1). Excluded types `4`/`5` (or
|
||||||
message and returns an empty tibble (v0.1 corpus excludes those types).}
|
`"special_district"` / `"school_district"`) trigger an explanatory
|
||||||
|
message and an empty result.}
|
||||||
}
|
}
|
||||||
\value{
|
\value{
|
||||||
Tibble from `canonical_fips_xwalk` sorted by `population_acs`
|
A tibble of `canonical_fips_xwalk` rows. In utility mode, all
|
||||||
descending (`NULL`s last).
|
matches sorted by `population_acs` desc. In basket mode, resolved
|
||||||
|
rows in input order, with `attr(., "resolution")` set to the
|
||||||
|
sidecar tibble.
|
||||||
}
|
}
|
||||||
\description{
|
\description{
|
||||||
Returns rows from `canonical_fips_xwalk` matching the supplied filters.
|
Resolves human-readable place names into rows of `canonical_fips_xwalk`,
|
||||||
Intended as the entry point users call to resolve a human-readable place
|
the cross-vintage canonical-government registry. Operates in two modes:
|
||||||
name into one or more `canonical_govid` values before calling
|
}
|
||||||
[cog_spending()] / [cog_revenue()] / etc.
|
\details{
|
||||||
|
* **Utility mode** (single `name`, the original behavior): returns all
|
||||||
|
rows whose `gov_name` matches the regex case-insensitively, sorted by
|
||||||
|
`population_acs` descending. Useful for exploratory lookups.
|
||||||
|
* **Basket mode** (`length(name) > 1`): resolves each input row to a
|
||||||
|
single canonical govid and returns a tibble in input order, suitable
|
||||||
|
for piping straight into [cog_spending()] / [cog_revenue()] /
|
||||||
|
[cog_geographic_rollup()]. Carries an audit sidecar accessible via
|
||||||
|
[cog_basket_resolution()] / [cog_basket_unresolved()].
|
||||||
|
|
||||||
|
|
||||||
|
**Basket-mode resolution algorithm** (per input row):
|
||||||
|
1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`.
|
||||||
|
2. **Exact pass:** case-insensitive equality against `gov_name`.
|
||||||
|
Single hit -> resolved. Multiple -> step 4.
|
||||||
|
3. **Substring fallback:** case-insensitive regex against `gov_name`.
|
||||||
|
Single hit -> resolved (`match_method = "substring"`). Zero hits ->
|
||||||
|
`status = "no_match"`. Multiple hits -> step 4.
|
||||||
|
4. **Disambiguation:** if matches share one `govs_type`, pick the
|
||||||
|
largest-population row (`status = "largest_pop"`). If they span >=2
|
||||||
|
types, no row is added (`status = "ambiguous"`); the user should
|
||||||
|
re-run with `type` specified.
|
||||||
|
|
||||||
|
Resolved rows form the returned tibble in input order. Unresolved
|
||||||
|
inputs (`ambiguous` / `no_match`) appear only in the sidecar.
|
||||||
|
}
|
||||||
|
\examples{
|
||||||
|
\dontrun{
|
||||||
|
# Utility mode — exploratory regex lookup
|
||||||
|
cog_gov_search("broward", state = "FL")
|
||||||
|
|
||||||
|
# Basket mode — resolve a known cohort
|
||||||
|
basket <- cog_gov_search(
|
||||||
|
name = c("BROWARD COUNTY", "SAN DIEGO CITY", "AUSTIN CITY"),
|
||||||
|
state = c("FL", "CA", "TX")
|
||||||
|
)
|
||||||
|
basket
|
||||||
|
|
||||||
|
# Inspect resolution audit
|
||||||
|
cog_basket_resolution(basket)
|
||||||
|
|
||||||
|
# Pipe into a spending query
|
||||||
|
library(dplyr)
|
||||||
|
basket |> cog_spending(years = 2019:2020, category = "Police")
|
||||||
|
|
||||||
|
# Iteratively refine ambiguous matches
|
||||||
|
partial <- cog_gov_search(
|
||||||
|
name = c("Broward", "San Diego"), # San Diego is ambiguous
|
||||||
|
state = c("FL", "CA")
|
||||||
|
)
|
||||||
|
cog_basket_unresolved(partial)
|
||||||
|
refined <- cog_gov_search(
|
||||||
|
name = c("Broward", "San Diego"),
|
||||||
|
state = c("FL", "CA"),
|
||||||
|
type = c(NA, "city") # disambiguate
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
\seealso{
|
||||||
|
[cog_basket_resolution()], [cog_basket_unresolved()],
|
||||||
|
[cog_spending()], [cog_revenue()].
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,51 @@
|
|||||||
|
test_that("cog_basket_resolution returns the sidecar tibble", {
|
||||||
|
basket <- suppressMessages(cog_gov_search(
|
||||||
|
name = c("Broward", "Notarealplace"),
|
||||||
|
state = c("FL", "NY")
|
||||||
|
))
|
||||||
|
res <- cog_basket_resolution(basket)
|
||||||
|
expect_s3_class(res, "tbl_df")
|
||||||
|
expect_equal(nrow(res), 2L)
|
||||||
|
expect_setequal(colnames(res), c(
|
||||||
|
"query_name", "query_state", "query_type", "status",
|
||||||
|
"match_method", "canonical_govid", "gov_name", "n_candidates"
|
||||||
|
))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_basket_resolution(expand_candidates = TRUE) keeps candidates list-col", {
|
||||||
|
basket <- suppressMessages(cog_gov_search(
|
||||||
|
name = c("Broward", "Notarealplace"),
|
||||||
|
state = c("FL", "NY")
|
||||||
|
))
|
||||||
|
res <- cog_basket_resolution(basket, expand_candidates = TRUE)
|
||||||
|
expect_true("candidates" %in% colnames(res))
|
||||||
|
expect_true(is.list(res$candidates))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_basket_resolution errors on a non-basket tibble", {
|
||||||
|
utility <- cog_gov_search("BROWARD")
|
||||||
|
expect_error(
|
||||||
|
cog_basket_resolution(utility),
|
||||||
|
regexp = "no resolution attribute"
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_basket_unresolved filters to ambiguous and no_match", {
|
||||||
|
basket <- suppressMessages(cog_gov_search(
|
||||||
|
name = c("Broward", "San Diego", "Notarealplace"),
|
||||||
|
state = c("FL", "CA", "NY")
|
||||||
|
))
|
||||||
|
unres <- cog_basket_unresolved(basket)
|
||||||
|
expect_equal(nrow(unres), 2L)
|
||||||
|
expect_setequal(unres$status, c("ambiguous", "no_match"))
|
||||||
|
expect_true("candidates" %in% colnames(unres))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_basket_unresolved returns 0 rows when basket is clean", {
|
||||||
|
basket <- cog_gov_search(
|
||||||
|
name = c("BROWARD COUNTY", "SAN DIEGO CITY"),
|
||||||
|
state = c("FL", "CA")
|
||||||
|
)
|
||||||
|
unres <- cog_basket_unresolved(basket)
|
||||||
|
expect_equal(nrow(unres), 0L)
|
||||||
|
})
|
||||||
@@ -72,3 +72,366 @@ test_that("cog_gov_search with no filters returns full registry", {
|
|||||||
r <- cog_gov_search()
|
r <- cog_gov_search()
|
||||||
expect_gt(nrow(r), 1000L)
|
expect_gt(nrow(r), 1000L)
|
||||||
})
|
})
|
||||||
|
|
||||||
|
# ---- basket mode internal helpers ----
|
||||||
|
|
||||||
|
test_that(".validate_basket_args recycles state from length 1", {
|
||||||
|
out <- uscogdata:::.validate_basket_args(
|
||||||
|
name = c("Broward", "San Diego", "Austin"),
|
||||||
|
state = "FL",
|
||||||
|
type = NULL
|
||||||
|
)
|
||||||
|
expect_equal(out$name, c("Broward", "San Diego", "Austin"))
|
||||||
|
expect_equal(out$state, c("FL", "FL", "FL"))
|
||||||
|
expect_equal(out$type, c(NA_character_, NA_character_, NA_character_))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".validate_basket_args recycles type from length 1", {
|
||||||
|
out <- uscogdata:::.validate_basket_args(
|
||||||
|
name = c("San Diego", "Oakland"),
|
||||||
|
state = "CA",
|
||||||
|
type = "city"
|
||||||
|
)
|
||||||
|
expect_equal(out$type, c("city", "city"))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".validate_basket_args accepts per-row state and type", {
|
||||||
|
out <- uscogdata:::.validate_basket_args(
|
||||||
|
name = c("Broward", "San Diego"),
|
||||||
|
state = c("FL", "CA"),
|
||||||
|
type = c(NA, "city")
|
||||||
|
)
|
||||||
|
expect_equal(out$state, c("FL", "CA"))
|
||||||
|
expect_equal(out$type, c(NA_character_, "city"))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".validate_basket_args rejects length-mismatched state", {
|
||||||
|
expect_error(
|
||||||
|
uscogdata:::.validate_basket_args(
|
||||||
|
name = c("Broward", "San Diego", "Austin"),
|
||||||
|
state = c("FL", "CA"),
|
||||||
|
type = NULL
|
||||||
|
),
|
||||||
|
regexp = "must be length 1 or 3"
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".validate_basket_args rejects length-mismatched type", {
|
||||||
|
expect_error(
|
||||||
|
uscogdata:::.validate_basket_args(
|
||||||
|
name = c("Broward", "San Diego"),
|
||||||
|
state = "FL",
|
||||||
|
type = c("county", "city", "city")
|
||||||
|
),
|
||||||
|
regexp = "must be length 1 or 2"
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".validate_basket_args allows NULL state and type", {
|
||||||
|
out <- uscogdata:::.validate_basket_args(
|
||||||
|
name = c("Broward", "San Diego"),
|
||||||
|
state = NULL,
|
||||||
|
type = NULL
|
||||||
|
)
|
||||||
|
expect_equal(out$state, c(NA_character_, NA_character_))
|
||||||
|
expect_equal(out$type, c(NA_character_, NA_character_))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".resolve_basket_row exact match returns one row", {
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
out <- uscogdata:::.resolve_basket_row(
|
||||||
|
name = "BROWARD COUNTY", state = "FL", type = NA_character_, con = con
|
||||||
|
)
|
||||||
|
expect_equal(out$status, "resolved")
|
||||||
|
expect_equal(out$match_method, "exact")
|
||||||
|
expect_equal(out$n_candidates, 1L)
|
||||||
|
expect_equal(nrow(out$row), 1L)
|
||||||
|
expect_equal(out$row$canonical_govid, "101006006")
|
||||||
|
expect_equal(out$row$gov_name, "BROWARD COUNTY")
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".resolve_basket_row exact match is case-insensitive", {
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
out <- uscogdata:::.resolve_basket_row(
|
||||||
|
name = "broward county", state = "FL", type = NA_character_, con = con
|
||||||
|
)
|
||||||
|
expect_equal(out$status, "resolved")
|
||||||
|
expect_equal(out$match_method, "exact")
|
||||||
|
expect_equal(out$row$canonical_govid, "101006006")
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".resolve_basket_row exact match honors per-row type", {
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
out <- uscogdata:::.resolve_basket_row(
|
||||||
|
name = "SAN DIEGO CITY", state = "CA", type = "city", con = con
|
||||||
|
)
|
||||||
|
expect_equal(out$status, "resolved")
|
||||||
|
expect_equal(out$row$canonical_govid, "052037010")
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".resolve_basket_row substring fallback resolves single match", {
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
out <- uscogdata:::.resolve_basket_row(
|
||||||
|
name = "Broward", state = "FL", type = NA_character_, con = con
|
||||||
|
)
|
||||||
|
expect_equal(out$status, "resolved")
|
||||||
|
expect_equal(out$match_method, "substring")
|
||||||
|
expect_equal(out$n_candidates, 1L)
|
||||||
|
expect_equal(out$row$canonical_govid, "101006006")
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".resolve_basket_row no_match returns 0-row tibble", {
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
out <- uscogdata:::.resolve_basket_row(
|
||||||
|
name = "Notarealplace", state = "NY", type = NA_character_, con = con
|
||||||
|
)
|
||||||
|
expect_equal(out$status, "no_match")
|
||||||
|
expect_true(is.na(out$match_method))
|
||||||
|
expect_equal(out$n_candidates, 0L)
|
||||||
|
expect_equal(nrow(out$row), 0L)
|
||||||
|
expect_equal(nrow(out$candidates), 0L)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".resolve_basket_row treats empty/whitespace name as no_match", {
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
out_empty <- uscogdata:::.resolve_basket_row(
|
||||||
|
name = "", state = "FL", type = NA_character_, con = con
|
||||||
|
)
|
||||||
|
expect_equal(out_empty$status, "no_match")
|
||||||
|
|
||||||
|
out_ws <- uscogdata:::.resolve_basket_row(
|
||||||
|
name = " ", state = "FL", type = NA_character_, con = con
|
||||||
|
)
|
||||||
|
expect_equal(out_ws$status, "no_match")
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".resolve_basket_row largest_pop within single type", {
|
||||||
|
# FL Miami substring matches 10 cities (all govs_type = 2), largest pop
|
||||||
|
# is MIAMI CITY at 443665.
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
out <- uscogdata:::.resolve_basket_row(
|
||||||
|
name = "Miami", state = "FL", type = NA_character_, con = con
|
||||||
|
)
|
||||||
|
expect_equal(out$status, "largest_pop")
|
||||||
|
expect_equal(out$match_method, "substring")
|
||||||
|
expect_gte(out$n_candidates, 2L)
|
||||||
|
expect_equal(out$row$canonical_govid, "102013013")
|
||||||
|
expect_equal(out$row$gov_name, "MIAMI CITY")
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".resolve_basket_row ambiguous across types", {
|
||||||
|
# SAN DIEGO substring matches both SAN DIEGO COUNTY (type 1) and
|
||||||
|
# SAN DIEGO CITY (type 2) in CA.
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
out <- uscogdata:::.resolve_basket_row(
|
||||||
|
name = "San Diego", state = "CA", type = NA_character_, con = con
|
||||||
|
)
|
||||||
|
expect_equal(out$status, "ambiguous")
|
||||||
|
expect_true(is.na(out$match_method))
|
||||||
|
expect_equal(out$n_candidates, 2L)
|
||||||
|
expect_equal(nrow(out$row), 0L)
|
||||||
|
expect_equal(nrow(out$candidates), 2L)
|
||||||
|
expect_setequal(out$candidates$govs_type, c(1L, 2L))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".resolve_basket_row resolves with type override on ambiguous case", {
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
out <- uscogdata:::.resolve_basket_row(
|
||||||
|
name = "San Diego", state = "CA", type = "city", con = con
|
||||||
|
)
|
||||||
|
expect_equal(out$status, "resolved")
|
||||||
|
expect_equal(out$match_method, "substring")
|
||||||
|
expect_equal(out$row$canonical_govid, "052037010")
|
||||||
|
})
|
||||||
|
|
||||||
|
# ---- basket mode public surface ----
|
||||||
|
|
||||||
|
test_that("cog_gov_search basket mode resolves clean inputs in input order", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
basket <- cog_gov_search(
|
||||||
|
name = c("BROWARD COUNTY", "SAN DIEGO CITY", "AUSTIN CITY"),
|
||||||
|
state = c("FL", "CA", "TX")
|
||||||
|
)
|
||||||
|
expect_s3_class(basket, "tbl_df")
|
||||||
|
expect_equal(nrow(basket), 3L)
|
||||||
|
expect_equal(basket$canonical_govid, c("101006006", "052037010", "442227001"))
|
||||||
|
expect_equal(basket$gov_name, c("BROWARD COUNTY", "SAN DIEGO CITY", "AUSTIN CITY"))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_gov_search basket mode attaches a resolution sidecar", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
basket <- cog_gov_search(
|
||||||
|
name = c("Broward", "San Diego"),
|
||||||
|
state = c("FL", "CA"),
|
||||||
|
type = c(NA, "city")
|
||||||
|
)
|
||||||
|
res <- attr(basket, "resolution")
|
||||||
|
expect_s3_class(res, "tbl_df")
|
||||||
|
expect_equal(nrow(res), 2L)
|
||||||
|
expect_equal(res$query_name, c("Broward", "San Diego"))
|
||||||
|
expect_equal(res$query_state, c("FL", "CA"))
|
||||||
|
expect_equal(res$query_type, c(NA_character_, "city"))
|
||||||
|
expect_equal(res$status, c("resolved", "resolved"))
|
||||||
|
expect_equal(res$match_method, c("substring", "substring"))
|
||||||
|
expect_true(is.list(res$candidates))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_gov_search basket mode skips ambiguous and no_match rows", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
basket <- suppressMessages(cog_gov_search(
|
||||||
|
name = c("Broward", "San Diego", "Notarealplace"),
|
||||||
|
state = c("FL", "CA", "NY")
|
||||||
|
))
|
||||||
|
# Broward resolves; San Diego ambiguous; Notarealplace no_match.
|
||||||
|
expect_equal(nrow(basket), 1L)
|
||||||
|
expect_equal(basket$canonical_govid, "101006006")
|
||||||
|
res <- attr(basket, "resolution")
|
||||||
|
expect_equal(nrow(res), 3L)
|
||||||
|
expect_equal(res$status, c("resolved", "ambiguous", "no_match"))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_gov_search basket mode preserves input order", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
basket <- cog_gov_search(
|
||||||
|
name = c("AUSTIN CITY", "BROWARD COUNTY", "SAN DIEGO CITY"),
|
||||||
|
state = c("TX", "FL", "CA")
|
||||||
|
)
|
||||||
|
expect_equal(basket$gov_name, c("AUSTIN CITY", "BROWARD COUNTY", "SAN DIEGO CITY"))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_gov_search basket mode recycles single state", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
basket <- cog_gov_search(
|
||||||
|
name = c("SAN DIEGO CITY", "OAKLAND CITY"),
|
||||||
|
state = "CA"
|
||||||
|
)
|
||||||
|
expect_equal(nrow(basket), 2L)
|
||||||
|
expect_equal(basket$canonical_govid, c("052037010", "052001009"))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_gov_search basket mode within-type largest_pop records candidates", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
basket <- suppressMessages(cog_gov_search(
|
||||||
|
name = c("Miami", "OAKLAND CITY"),
|
||||||
|
state = c("FL", "CA")
|
||||||
|
))
|
||||||
|
expect_equal(nrow(basket), 2L)
|
||||||
|
res <- attr(basket, "resolution")
|
||||||
|
miami_row <- res[res$query_name == "Miami", ]
|
||||||
|
expect_equal(miami_row$status, "largest_pop")
|
||||||
|
expect_equal(miami_row$canonical_govid, "102013013")
|
||||||
|
expect_gte(miami_row$n_candidates, 2L)
|
||||||
|
expect_gte(nrow(miami_row$candidates[[1]]), 2L)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_gov_search utility mode (length-1 name) has no sidecar", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
r <- cog_gov_search("BROWARD")
|
||||||
|
expect_null(attr(r, "resolution"))
|
||||||
|
expect_gte(nrow(r), 1L)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_gov_search basket mode validates argument lengths", {
|
||||||
|
expect_error(
|
||||||
|
cog_gov_search(
|
||||||
|
name = c("Broward", "San Diego", "Austin"),
|
||||||
|
state = c("FL", "CA")
|
||||||
|
),
|
||||||
|
regexp = "must be length 1 or 3"
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
# ---- basket mode summary message ----
|
||||||
|
|
||||||
|
test_that("cog_gov_search basket mode is silent on clean basket", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
expect_message(
|
||||||
|
cog_gov_search(
|
||||||
|
name = c("BROWARD COUNTY", "SAN DIEGO CITY"),
|
||||||
|
state = c("FL", "CA")
|
||||||
|
),
|
||||||
|
regexp = NA # NA = expect no message
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_gov_search basket mode reports breakdown on partial basket", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
expect_message(
|
||||||
|
cog_gov_search(
|
||||||
|
name = c("Broward", "San Diego", "Notarealplace"),
|
||||||
|
state = c("FL", "CA", "NY")
|
||||||
|
),
|
||||||
|
regexp = "Basket resolved 1 of 3"
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_gov_search basket mode message points to the sidecar accessor", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
expect_message(
|
||||||
|
cog_gov_search(
|
||||||
|
name = c("Broward", "Notarealplace"),
|
||||||
|
state = c("FL", "NY")
|
||||||
|
),
|
||||||
|
regexp = "cog_basket_resolution"
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
# ---- F1: per-row excluded type soft-fail ----
|
||||||
|
|
||||||
|
test_that(".resolve_basket_row treats excluded type as no_match (not abort)", {
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
out <- uscogdata:::.resolve_basket_row(
|
||||||
|
name = "Some District", state = "CA", type = "special_district", con = con
|
||||||
|
)
|
||||||
|
expect_equal(out$status, "no_match")
|
||||||
|
expect_true(is.na(out$match_method))
|
||||||
|
expect_equal(out$n_candidates, 0L)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_gov_search basket mode skips per-row excluded type without aborting", {
|
||||||
|
basket <- suppressMessages(cog_gov_search(
|
||||||
|
name = c("BROWARD COUNTY", "Some District"),
|
||||||
|
state = c("FL", "FL"),
|
||||||
|
type = c(NA, "special_district")
|
||||||
|
))
|
||||||
|
# Broward should resolve; the special_district row should be no_match.
|
||||||
|
expect_equal(nrow(basket), 1L)
|
||||||
|
expect_equal(basket$canonical_govid, "101006006")
|
||||||
|
res <- attr(basket, "resolution")
|
||||||
|
expect_equal(res$status, c("resolved", "no_match"))
|
||||||
|
# query_type should record what the user passed for the excluded-type row
|
||||||
|
expect_equal(res$query_type, c(NA_character_, "special_district"))
|
||||||
|
})
|
||||||
|
|
||||||
|
# ---- F2: malformed regex name soft-fail ----
|
||||||
|
|
||||||
|
test_that(".resolve_basket_row treats malformed regex name as no_match", {
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
# Unbalanced parens would be a regex parse error if not escaped.
|
||||||
|
out <- uscogdata:::.resolve_basket_row(
|
||||||
|
name = "San(Diego", state = "CA", type = NA_character_, con = con
|
||||||
|
)
|
||||||
|
expect_equal(out$status, "no_match")
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".resolve_basket_row escapes regex metacharacters in name", {
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
# Confirm that names with various metacharacters don't error.
|
||||||
|
expect_no_error(uscogdata:::.resolve_basket_row(
|
||||||
|
name = "Foo*Bar+Baz", state = "FL", type = NA_character_, con = con
|
||||||
|
))
|
||||||
|
})
|
||||||
|
|
||||||
|
# ---- F3: all-no-match basket public surface ----
|
||||||
|
|
||||||
|
test_that("cog_gov_search basket all-no-match returns 0-row tibble with full sidecar", {
|
||||||
|
basket <- suppressMessages(cog_gov_search(
|
||||||
|
name = c("Notarealplace1", "Notarealplace2"),
|
||||||
|
state = c("NY", "CA")
|
||||||
|
))
|
||||||
|
expect_equal(nrow(basket), 0L)
|
||||||
|
expect_true("canonical_govid" %in% names(basket))
|
||||||
|
res <- attr(basket, "resolution")
|
||||||
|
expect_equal(nrow(res), 2L)
|
||||||
|
expect_true(all(res$status == "no_match"))
|
||||||
|
})
|
||||||
|
|||||||
@@ -116,3 +116,15 @@ test_that("cog_spending rejects data.frame without canonical_govid column", {
|
|||||||
bad <- tibble::tibble(foo = "bar")
|
bad <- tibble::tibble(foo = "bar")
|
||||||
expect_error(cog_spending(bad, 2020L), "canonical_govid")
|
expect_error(cog_spending(bad, 2020L), "canonical_govid")
|
||||||
})
|
})
|
||||||
|
|
||||||
|
test_that("cog_spending accepts a basket-mode cog_gov_search result", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
basket <- cog_gov_search(
|
||||||
|
name = c("BROWARD COUNTY", "SAN DIEGO COUNTY"),
|
||||||
|
state = c("FL", "CA")
|
||||||
|
)
|
||||||
|
expect_equal(nrow(basket), 2L)
|
||||||
|
spending <- cog_spending(basket, years = 2019:2020, category = "Police")
|
||||||
|
expect_s3_class(spending, "tbl_df")
|
||||||
|
expect_setequal(unique(spending$canonical_govid), basket$canonical_govid)
|
||||||
|
})
|
||||||
|
|||||||
Reference in New Issue
Block a user