Files
uscogdata/tests/testthat/test-gov-search-literal-match.R
T
jared 9233c3d18e
R-CMD-check / check (push) Successful in 3m3s
R-CMD-check / check (pull_request) Successful in 2m58s
test: add failing tests for Madison walkthrough findings
Six skipped tests, one per issue opened from the Madison walkthrough audit
(docs/walkthroughs/FINDINGS.md in cog_explorer). Each asserts the desired
behaviour, so it fails today and goes green when the fix lands; each is
guarded by a single skip() naming its issue and finding IDs, so the suite
stays green and activating a test is a one-line deletion.

  test-expenditure-concepts.R             #11  F-012, F-017, F-018
  test-revenue-concept-insurance-trust.R  #12  F-014
  test-coverage-disclosure.R              #13  F-020, F-023
  test-peer-summary-scope.R               #14  F-021
  test-amount-units-documented.R          #15  F-004
  test-gov-search-literal-match.R         #16  F-025

helper-walkthrough-raw.R reads the corpus's long parquet partitions directly,
bypassing uscogdata's SQL views. Every expected amount comes from there rather
than from the verb under test - verifying an absence through the filter that
creates it proves nothing, which was the most common defect in the audit itself.

Verified: with the skips removed all six fail (or error) against the bundled
fixture; with them in place the full suite is 576 pass / 0 fail / 6 skip.
2026-07-29 00:14:11 -04:00

59 lines
3.2 KiB
R

# Madison walkthrough audit -- finding F-025. Tracked as uscogdata#16.
# See docs/walkthroughs/FINDINGS.md in cog_explorer.
#
# cog_gov_search()'s UTILITY mode interpolates `name` into
# regexp_matches(gov_name, <name>, 'i')
# unescaped (R/search.R:102), while BASKET mode in the same file already routes
# it through .escape_regex() (R/search.R:307) with the comment "so `name` is
# treated as a literal substring". Two failure modes result:
# correctness -- a real government cannot be found by its own exact name, and
# a single "." matches everything (HTTP 200 both ways via the API);
# robustness -- malformed regex reaches the engine and errors, which cog-api
# surfaces as a 500, reachable by typing a real name one
# character at a time.
#
# NOT asserted here: the finding's `q=St. Louis` example. Under correct literal
# matching that search still returns 0 rows, because the stored name is
# "ST LOUIS CITY" with no period -- it demonstrates today's over-matching
# semantics, not a row the fix makes findable.
test_that("cog_gov_search() matches name literally, not as an unescaped regex", {
testthat::skip("Blocked on uscogdata#16 (finding F-025)")
# -- correctness (1): a government must be findable by its own exact name ---
# FREDONIA (BRISCOE) CITY is real; today the parentheses are read as regex
# grouping, so its own complete name matches nothing.
fredonia <- cog_gov_search(name = "FREDONIA (BRISCOE) CITY")
expect_equal(nrow(fredonia), 1L)
expect_equal(fredonia$canonical_govid, "052117184386")
expect_equal(cog_gov_search(name = "FREDONIA (BRISCOE)")$canonical_govid,
"052117184386")
# -- correctness (2): a metacharacter must not become a wildcard ------------
# No Wisconsin city or village name contains a literal period -- established
# against the raw registry below, NOT through the verb under test. A literal
# search for "." must therefore return nothing; today it returns all 608.
con <- DBI::dbConnect(duckdb::duckdb())
on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE)
xwalk <- paste0(sub("/$", "", Sys.getenv("USCOGDATA_URL")),
"/data/canonical_fips_xwalk.parquet")
with_dot <- DBI::dbGetQuery(con, paste0(
"SELECT COUNT(*) n FROM read_parquet('", xwalk, "') ",
"WHERE fips_state = '55' AND govs_type = 2 AND gov_name LIKE '%.%'"))
expect_equal(as.integer(with_dot$n[[1]]), 0L)
expect_equal(nrow(cog_gov_search(name = ".", state = "WI", type = "city")), 0L)
expect_equal(nrow(cog_gov_search(name = "M.dison", state = "WI", type = "city")), 0L)
expect_equal(nrow(cog_gov_search(name = "Mad(i|o)son", state = "WI", type = "city")), 0L)
# A metacharacter-free name still resolves exactly as before.
expect_equal(nrow(cog_gov_search(name = "Madison", state = "WI", type = "city")), 1L)
# -- robustness: malformed pattern text returns no rows, and does not error --
# "[" alone, and "Athens-Clarke County (bal" -- an in-progress substring of
# ATHENS-CLARKE COUNTY (BALANCE), a real government -- both currently raise
# (DuckDB: "Invalid Input Error: missing ]").
expect_equal(nrow(cog_gov_search(name = "[")), 0L)
expect_equal(nrow(cog_gov_search(name = "Athens-Clarke County (bal")), 0L)
})