test: re-baseline canonical_govid literals to 12-char namespace

Swaps every hardcoded 9-char canonical_govid literal (Broward County,
Fort Lauderdale City, Florida/Alabama state govts, Bexar/Tarrant/Wayne
counties, San Diego/Oakland/Miami/Austin cities) for its 12-char Phase P
equivalent, resolved by name+type+state against the regenerated fixture
xwalk. Also updates two gov_name search patterns that no longer match
under Phase P canonical naming ("FLORIDA STATE GOVT" -> "FLORIDA"; the
"Miami" substring test now pins type = "city" since MIAMI-DADE COUNTY's
canonical name now also contains "Miami", which would otherwise make the
match ambiguous across govs_types instead of resolving via largest-pop).
Underlying per-year population figures for Broward County and Alabama
are unchanged, so no expected data-value literals needed recomputation.
Suite: 126 test blocks / 336 expectations, 0 FAIL / 0 WARN / 0 SKIP.
This commit is contained in:
2026-07-11 09:32:02 -04:00
parent 570a9408a2
commit 92c9a7382e
9 changed files with 95 additions and 85 deletions
+5 -5
View File
@@ -1,6 +1,6 @@
test_that("cog_explain prints verb header and target", { test_that("cog_explain prints verb header and target", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_spending("101006006", 2020L, "Corrections") r <- cog_spending("121011212191", 2020L, "Corrections")
# cli writes to stderr; capture both stdout and message streams. # cli writes to stderr; capture both stdout and message streams.
txt <- paste(c( txt <- paste(c(
capture.output(cog_explain(r)), capture.output(cog_explain(r)),
@@ -8,19 +8,19 @@ test_that("cog_explain prints verb header and target", {
), collapse = "\n") ), collapse = "\n")
expect_true(grepl("cog_spending", txt)) expect_true(grepl("cog_spending", txt))
expect_true(grepl("Corrections", txt)) expect_true(grepl("Corrections", txt))
expect_true(grepl("101006006", txt)) expect_true(grepl("121011212191", txt))
}) })
test_that("cog_explain format='list' returns structured provenance", { test_that("cog_explain format='list' returns structured provenance", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_spending("101006006", 2020L, "Corrections") r <- cog_spending("121011212191", 2020L, "Corrections")
prov <- cog_explain(r, format = "list") prov <- cog_explain(r, format = "list")
expect_identical(prov, attr(r, "provenance")) expect_identical(prov, attr(r, "provenance"))
}) })
test_that("cog_explain returns result invisibly for chaining", { test_that("cog_explain returns result invisibly for chaining", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_spending("101006006", 2020L, "Corrections") r <- cog_spending("121011212191", 2020L, "Corrections")
res <- withVisible(cog_explain(r)) res <- withVisible(cog_explain(r))
expect_false(res$visible) expect_false(res$visible)
expect_identical(res$value, r) expect_identical(res$value, r)
@@ -34,7 +34,7 @@ test_that("cog_explain errors on non-verb input", {
test_that("cog_explain prints denominator + popyear_range + counts", { test_that("cog_explain prints denominator + popyear_range + counts", {
skip_if_no_corpus() skip_if_no_corpus()
with_fixture_corpus({ with_fixture_corpus({
r <- cog_spending("101006006", years = 2019:2020, r <- cog_spending("121011212191", years = 2019:2020,
category = "Police", per_capita = TRUE) category = "Police", per_capita = TRUE)
out <- paste(c( out <- paste(c(
capture.output(cog_explain(r)), capture.output(cog_explain(r)),
+2 -2
View File
@@ -61,7 +61,7 @@ test_that("cog_mirror reads back via a fresh session against the mirror", {
cog_close() cog_close()
options(uscogdata.url = paste0(normalizePath(tmp), "/")) options(uscogdata.url = paste0(normalizePath(tmp), "/"))
r <- cog_spending("101006006", 2020L, "Corrections") r <- cog_spending("121011212191", 2020L, "Corrections")
expect_gt(nrow(r), 0L) expect_gt(nrow(r), 0L)
expect_equal(unique(r$canonical_govid), "101006006") expect_equal(unique(r$canonical_govid), "121011212191")
}) })
+20 -20
View File
@@ -1,25 +1,25 @@
test_that("cog_find_peers returns same-type peers in the default pop band", { test_that("cog_find_peers returns same-type peers in the default pop band", {
skip_if_no_corpus() skip_if_no_corpus()
peers <- cog_find_peers("101006006") # Broward County peers <- cog_find_peers("121011212191") # Broward County
expect_s3_class(peers, "tbl_df") expect_s3_class(peers, "tbl_df")
expected_cols <- c("canonical_govid", "gov_name", "fips_state", expected_cols <- c("canonical_govid", "gov_name", "fips_state",
"population", "pop_ratio", "rank") "population", "pop_ratio", "rank")
expect_true(all(expected_cols %in% names(peers))) expect_true(all(expected_cols %in% names(peers)))
expect_true(all(peers$pop_ratio >= 0.7 & peers$pop_ratio <= 1.3)) expect_true(all(peers$pop_ratio >= 0.7 & peers$pop_ratio <= 1.3))
expect_false("101006006" %in% peers$canonical_govid) expect_false("121011212191" %in% peers$canonical_govid)
expect_equal(peers$rank, seq_len(nrow(peers))) expect_equal(peers$rank, seq_len(nrow(peers)))
}) })
test_that("cog_find_peers respects same_state restriction", { test_that("cog_find_peers respects same_state restriction", {
skip_if_no_corpus() skip_if_no_corpus()
peers <- cog_find_peers("101006006", same_state = TRUE, peers <- cog_find_peers("121011212191", same_state = TRUE,
pop_range = c(0.1, 10)) pop_range = c(0.1, 10))
expect_true(all(peers$fips_state == "12")) expect_true(all(peers$fips_state == "12"))
}) })
test_that("cog_find_peers absolute pop range works", { test_that("cog_find_peers absolute pop range works", {
skip_if_no_corpus() skip_if_no_corpus()
peers <- cog_find_peers("101006006", peers <- cog_find_peers("121011212191",
pop_range = c(1.5e6, 2.5e6), pop_range = c(1.5e6, 2.5e6),
is_ratio = FALSE, max_peers = 20L) is_ratio = FALSE, max_peers = 20L)
expect_true(all(peers$population >= 1.5e6 & expect_true(all(peers$population >= 1.5e6 &
@@ -33,8 +33,8 @@ test_that("cog_find_peers errors cleanly on unknown govid", {
test_that("cog_peer_compare accepts a cog_find_peers result directly", { test_that("cog_peer_compare accepts a cog_find_peers result directly", {
skip_if_no_corpus() skip_if_no_corpus()
peers <- cog_find_peers("101006006", max_peers = 4L) peers <- cog_find_peers("121011212191", max_peers = 4L)
r <- cog_peer_compare("101006006", peers, "Police", years = 2020L) r <- cog_peer_compare("121011212191", peers, "Police", years = 2020L)
expect_s3_class(r, "tbl_df") expect_s3_class(r, "tbl_df")
expect_true("role" %in% names(r)) expect_true("role" %in% names(r))
expect_setequal( expect_setequal(
@@ -47,8 +47,8 @@ test_that("cog_peer_compare accepts a cog_find_peers result directly", {
test_that("cog_peer_compare accepts a character vector of govids", { test_that("cog_peer_compare accepts a character vector of govids", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_peer_compare( r <- cog_peer_compare(
"101006006", "121011212191",
peers = c("441015015", "441220220"), # Bexar, Tarrant peers = c("481029175853", "481439135072"), # Bexar, Tarrant
category = "Police", years = 2020L category = "Police", years = 2020L
) )
expect_true("peer" %in% r$role) expect_true("peer" %in% r$role)
@@ -58,8 +58,8 @@ test_that("cog_peer_compare accepts a character vector of govids", {
test_that("cog_peer_compare summary rows use real per-capita when requested", { test_that("cog_peer_compare summary rows use real per-capita when requested", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_peer_compare( r <- cog_peer_compare(
"101006006", "121011212191",
peers = c("441015015", "441220220", "231082082"), peers = c("481029175853", "481439135072", "261163166615"),
category = "Police", years = 2019:2020, category = "Police", years = 2019:2020,
per_capita = TRUE, adjust_to_year = 2022L per_capita = TRUE, adjust_to_year = 2022L
) )
@@ -72,8 +72,8 @@ test_that("cog_peer_compare summary rows use real per-capita when requested", {
test_that("cog_peer_compare provenance reports the outer verb + peer count", { test_that("cog_peer_compare provenance reports the outer verb + peer count", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_peer_compare("101006006", r <- cog_peer_compare("121011212191",
peers = c("441015015", "441220220"), peers = c("481029175853", "481439135072"),
category = "Police", years = 2020L) category = "Police", years = 2020L)
prov <- attr(r, "provenance") prov <- attr(r, "provenance")
expect_equal(prov$verb, "cog_peer_compare") expect_equal(prov$verb, "cog_peer_compare")
@@ -82,7 +82,7 @@ test_that("cog_peer_compare provenance reports the outer verb + peer count", {
test_that("cog_peer_compare handles zero peers gracefully", { test_that("cog_peer_compare handles zero peers gracefully", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_peer_compare("101006006", r <- cog_peer_compare("121011212191",
peers = character(0), peers = character(0),
category = "Police", years = 2020L) category = "Police", years = 2020L)
expect_true(all(r$role == "target")) expect_true(all(r$role == "target"))
@@ -91,7 +91,7 @@ test_that("cog_peer_compare handles zero peers gracefully", {
test_that("cog_find_peers defaults `year` to most recent observed year for target", { test_that("cog_find_peers defaults `year` to most recent observed year for target", {
skip_if_no_corpus() skip_if_no_corpus()
peers <- cog_find_peers("101006006") peers <- cog_find_peers("121011212191")
expect_equal(attr(peers, "cohort_year"), 2020L) expect_equal(attr(peers, "cohort_year"), 2020L)
# Returned column is now `population`, not `population_acs` # Returned column is now `population`, not `population_acs`
expect_true("population" %in% names(peers)) expect_true("population" %in% names(peers))
@@ -100,23 +100,23 @@ test_that("cog_find_peers defaults `year` to most recent observed year for targe
test_that("cog_find_peers honors an explicit `year`", { test_that("cog_find_peers honors an explicit `year`", {
skip_if_no_corpus() skip_if_no_corpus()
peers <- cog_find_peers("101006006", year = 2019L) peers <- cog_find_peers("121011212191", year = 2019L)
expect_equal(attr(peers, "cohort_year"), 2019L) expect_equal(attr(peers, "cohort_year"), 2019L)
}) })
test_that("cog_find_peers errors when target has no observed pop in `year`", { test_that("cog_find_peers errors when target has no observed pop in `year`", {
skip_if_no_corpus() skip_if_no_corpus()
expect_error( expect_error(
cog_find_peers("101006006", year = 1999L), cog_find_peers("121011212191", year = 1999L),
"no observed population" "no observed population"
) )
}) })
test_that("cog_peer_compare stamps cohort_year from peers attribute", { test_that("cog_peer_compare stamps cohort_year from peers attribute", {
skip_if_no_corpus() skip_if_no_corpus()
peers <- cog_find_peers("101006006", year = 2019L, max_peers = 4L, peers <- cog_find_peers("121011212191", year = 2019L, max_peers = 4L,
pop_range = c(0.5, 1.5)) pop_range = c(0.5, 1.5))
r <- cog_peer_compare("101006006", peers, "Police", years = 2020L) r <- cog_peer_compare("121011212191", peers, "Police", years = 2020L)
expect_true("cohort_year" %in% names(r)) expect_true("cohort_year" %in% names(r))
expect_true(all(r$cohort_year == 2019L)) expect_true(all(r$cohort_year == 2019L))
prov <- attr(r, "provenance") prov <- attr(r, "provenance")
@@ -130,8 +130,8 @@ test_that("cog_peer_compare stamps cohort_year from peers attribute", {
test_that("cog_peer_compare cohort_year is NA for bare character peers", { test_that("cog_peer_compare cohort_year is NA for bare character peers", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_peer_compare( r <- cog_peer_compare(
"101006006", "121011212191",
peers = c("441015015", "441220220"), peers = c("481029175853", "481439135072"),
category = "Police", years = 2020L category = "Police", years = 2020L
) )
expect_true(all(is.na(r$cohort_year))) expect_true(all(is.na(r$cohort_year)))
+5 -5
View File
@@ -1,24 +1,24 @@
test_that("cog_revenue returns expected shape for Broward Property Tax 2020", { test_that("cog_revenue returns expected shape for Broward Property Tax 2020", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_revenue("101006006", years = 2020L, category = "Property Tax") r <- cog_revenue("121011212191", years = 2020L, category = "Property Tax")
expect_s3_class(r, "tbl_df") expect_s3_class(r, "tbl_df")
expected_cols <- c("year", "canonical_govid", "gov_name", "revenue_subtype", expected_cols <- c("year", "canonical_govid", "gov_name", "revenue_subtype",
"category", "amt_nominal", "codes_included", "category", "amt_nominal", "codes_included",
"aggregate_fallback", "notes") "aggregate_fallback", "notes")
expect_true(all(expected_cols %in% names(r))) expect_true(all(expected_cols %in% names(r)))
expect_equal(unique(r$canonical_govid), "101006006") expect_equal(unique(r$canonical_govid), "121011212191")
expect_equal(unique(r$year), 2020L) expect_equal(unique(r$year), 2020L)
}) })
test_that("cog_revenue with no category filter returns multiple categories", { test_that("cog_revenue with no category filter returns multiple categories", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_revenue("101006006", years = 2020L) r <- cog_revenue("121011212191", years = 2020L)
expect_gt(length(unique(r$category)), 1L) expect_gt(length(unique(r$category)), 1L)
}) })
test_that("cog_revenue with per_capita + adjust_to_year adds all columns", { test_that("cog_revenue with per_capita + adjust_to_year adds all columns", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_revenue("101006006", 2020L, r <- cog_revenue("121011212191", 2020L,
per_capita = TRUE, adjust_to_year = 2022L) per_capita = TRUE, adjust_to_year = 2022L)
expect_true(all(c("amt_nominal", "amt_real", expect_true(all(c("amt_nominal", "amt_real",
"amt_per_capita_nominal", "amt_per_capita_real") %in% "amt_per_capita_nominal", "amt_per_capita_real") %in%
@@ -27,7 +27,7 @@ test_that("cog_revenue with per_capita + adjust_to_year adds all columns", {
test_that("cog_revenue result has provenance attribute", { test_that("cog_revenue result has provenance attribute", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_revenue("101006006", 2020L) r <- cog_revenue("121011212191", 2020L)
prov <- attr(r, "provenance") prov <- attr(r, "provenance")
expect_equal(prov$verb, "cog_revenue") expect_equal(prov$verb, "cog_revenue")
expect_true(grepl("revenue_annotated", prov$sql_query)) expect_true(grepl("revenue_annotated", prov$sql_query))
+15 -15
View File
@@ -2,9 +2,9 @@ test_that("cog_geographic_rollup aggregates state + county + city layers", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_geographic_rollup( r <- cog_geographic_rollup(
govids = list( govids = list(
state = "100000000", # Florida state govt state = "120000226351", # Florida state govt
county = "101006006", # Broward County county = "121011212191", # Broward County
city = "102006004" # Fort Lauderdale City city = "122011161585" # Fort Lauderdale City
), ),
category = "Police", category = "Police",
years = 2019:2020 years = 2019:2020
@@ -23,7 +23,7 @@ test_that("cog_geographic_rollup aggregates state + county + city layers", {
test_that("cog_geographic_rollup respects per_capita + adjust_to_year", { test_that("cog_geographic_rollup respects per_capita + adjust_to_year", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_geographic_rollup( r <- cog_geographic_rollup(
govids = list(county = "101006006", city = "102006004"), govids = list(county = "121011212191", city = "122011161585"),
category = "Police", category = "Police",
years = 2020L, years = 2020L,
per_capita = TRUE, per_capita = TRUE,
@@ -43,8 +43,8 @@ test_that("cog_geographic_rollup respects per_capita + adjust_to_year", {
test_that("cog_geographic_rollup scope_notes describe each layer", { test_that("cog_geographic_rollup scope_notes describe each layer", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_geographic_rollup( r <- cog_geographic_rollup(
govids = list(state = "100000000", county = "101006006", govids = list(state = "120000226351", county = "121011212191",
city = "102006004"), city = "122011161585"),
category = "Police", years = 2020L category = "Police", years = 2020L
) )
state_notes <- unique(r$scope_note[r$layer == "state"]) state_notes <- unique(r$scope_note[r$layer == "state"])
@@ -58,7 +58,7 @@ test_that("cog_geographic_rollup scope_notes describe each layer", {
test_that("cog_geographic_rollup single-layer call works", { test_that("cog_geographic_rollup single-layer call works", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_geographic_rollup( r <- cog_geographic_rollup(
govids = list(county = c("101006006")), govids = list(county = c("121011212191")),
category = "Corrections", category = "Corrections",
years = 2020L years = 2020L
) )
@@ -69,7 +69,7 @@ test_that("cog_geographic_rollup single-layer call works", {
test_that("cog_geographic_rollup provenance reports the outer verb", { test_that("cog_geographic_rollup provenance reports the outer verb", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_geographic_rollup( r <- cog_geographic_rollup(
govids = list(state = "100000000", county = "101006006"), govids = list(state = "120000226351", county = "121011212191"),
category = "Police", years = 2020L category = "Police", years = 2020L
) )
prov <- attr(r, "provenance") prov <- attr(r, "provenance")
@@ -80,7 +80,7 @@ test_that("cog_geographic_rollup provenance reports the outer verb", {
test_that("cog_geographic_rollup accepts data.frames per layer", { test_that("cog_geographic_rollup accepts data.frames per layer", {
skip_if_no_corpus() skip_if_no_corpus()
fl_state <- cog_gov_search("^FLORIDA STATE GOVT$", type = "state") fl_state <- cog_gov_search("^FLORIDA$", type = "state")
broward <- cog_gov_search("^BROWARD COUNTY$", state = "FL", type = "county") broward <- cog_gov_search("^BROWARD COUNTY$", state = "FL", type = "county")
r <- cog_geographic_rollup( r <- cog_geographic_rollup(
govids = list(state = fl_state, county = broward), govids = list(state = fl_state, county = broward),
@@ -92,9 +92,9 @@ test_that("cog_geographic_rollup accepts data.frames per layer", {
test_that("cog_geographic_rollup rejects invalid inputs", { test_that("cog_geographic_rollup rejects invalid inputs", {
expect_error(cog_geographic_rollup(list(), "Police", 2020L), "length") expect_error(cog_geographic_rollup(list(), "Police", 2020L), "length")
expect_error(cog_geographic_rollup(c("101006006"), "Police", 2020L), "list") expect_error(cog_geographic_rollup(c("121011212191"), "Police", 2020L), "list")
expect_error( expect_error(
cog_geographic_rollup(list(planet = "100000000"), "Police", 2020L), cog_geographic_rollup(list(planet = "120000226351"), "Police", 2020L),
"state|county|city" "state|county|city"
) )
}) })
@@ -103,8 +103,8 @@ test_that("cog_geographic_rollup per-capita uses summed per-year populations", {
skip_if_no_corpus() skip_if_no_corpus()
with_fixture_corpus({ with_fixture_corpus({
r <- cog_geographic_rollup( r <- cog_geographic_rollup(
govids = list(state = "010000000", govids = list(state = "010000226085",
county = "101006006"), county = "121011212191"),
category = "Police", category = "Police",
years = 2019:2020, years = 2019:2020,
per_capita = TRUE per_capita = TRUE
@@ -125,14 +125,14 @@ test_that("cog_geographic_rollup records included/excluded govids in provenance"
skip_if_no_corpus() skip_if_no_corpus()
with_fixture_corpus({ with_fixture_corpus({
r <- cog_geographic_rollup( r <- cog_geographic_rollup(
govids = list(county = "101006006"), govids = list(county = "121011212191"),
category = "Police", category = "Police",
years = 2019:2020, years = 2019:2020,
per_capita = TRUE per_capita = TRUE
) )
prov <- attr(r, "provenance") prov <- attr(r, "provenance")
expect_true("rollup" %in% names(prov)) expect_true("rollup" %in% names(prov))
expect_true("101006006" %in% prov$rollup$included_govids) expect_true("121011212191" %in% prov$rollup$included_govids)
expect_true(is.character(prov$rollup$excluded_govids)) expect_true(is.character(prov$rollup$excluded_govids))
}) })
}) })
+22 -14
View File
@@ -146,7 +146,7 @@ test_that(".resolve_basket_row exact match returns one row", {
expect_equal(out$match_method, "exact") expect_equal(out$match_method, "exact")
expect_equal(out$n_candidates, 1L) expect_equal(out$n_candidates, 1L)
expect_equal(nrow(out$row), 1L) expect_equal(nrow(out$row), 1L)
expect_equal(out$row$canonical_govid, "101006006") expect_equal(out$row$canonical_govid, "121011212191")
expect_equal(out$row$gov_name, "BROWARD COUNTY") expect_equal(out$row$gov_name, "BROWARD COUNTY")
}) })
@@ -157,7 +157,7 @@ test_that(".resolve_basket_row exact match is case-insensitive", {
) )
expect_equal(out$status, "resolved") expect_equal(out$status, "resolved")
expect_equal(out$match_method, "exact") expect_equal(out$match_method, "exact")
expect_equal(out$row$canonical_govid, "101006006") expect_equal(out$row$canonical_govid, "121011212191")
}) })
test_that(".resolve_basket_row exact match honors per-row type", { test_that(".resolve_basket_row exact match honors per-row type", {
@@ -166,7 +166,7 @@ test_that(".resolve_basket_row exact match honors per-row type", {
name = "SAN DIEGO CITY", state = "CA", type = "city", con = con name = "SAN DIEGO CITY", state = "CA", type = "city", con = con
) )
expect_equal(out$status, "resolved") expect_equal(out$status, "resolved")
expect_equal(out$row$canonical_govid, "052037010") expect_equal(out$row$canonical_govid, "062073207598")
}) })
test_that(".resolve_basket_row substring fallback resolves single match", { test_that(".resolve_basket_row substring fallback resolves single match", {
@@ -177,7 +177,7 @@ test_that(".resolve_basket_row substring fallback resolves single match", {
expect_equal(out$status, "resolved") expect_equal(out$status, "resolved")
expect_equal(out$match_method, "substring") expect_equal(out$match_method, "substring")
expect_equal(out$n_candidates, 1L) expect_equal(out$n_candidates, 1L)
expect_equal(out$row$canonical_govid, "101006006") expect_equal(out$row$canonical_govid, "121011212191")
}) })
test_that(".resolve_basket_row no_match returns 0-row tibble", { test_that(".resolve_basket_row no_match returns 0-row tibble", {
@@ -207,15 +207,19 @@ test_that(".resolve_basket_row treats empty/whitespace name as no_match", {
test_that(".resolve_basket_row largest_pop within single type", { test_that(".resolve_basket_row largest_pop within single type", {
# FL Miami substring matches 10 cities (all govs_type = 2), largest pop # FL Miami substring matches 10 cities (all govs_type = 2), largest pop
# is MIAMI CITY at 443665. # is MIAMI CITY at 443665. Under Phase P canonical naming, MIAMI-DADE
# COUNTY (govs_type = 1) also contains "Miami", so `type = "city"` pins
# the match set to a single type (as the query docs promise it will for
# per-row `type`), keeping this test's original intent: multiple
# same-type name matches resolve to the largest-population row.
con <- uscogdata:::.ensure_session() con <- uscogdata:::.ensure_session()
out <- uscogdata:::.resolve_basket_row( out <- uscogdata:::.resolve_basket_row(
name = "Miami", state = "FL", type = NA_character_, con = con name = "Miami", state = "FL", type = "city", con = con
) )
expect_equal(out$status, "largest_pop") expect_equal(out$status, "largest_pop")
expect_equal(out$match_method, "substring") expect_equal(out$match_method, "substring")
expect_gte(out$n_candidates, 2L) expect_gte(out$n_candidates, 2L)
expect_equal(out$row$canonical_govid, "102013013") expect_equal(out$row$canonical_govid, "122086194757")
expect_equal(out$row$gov_name, "MIAMI CITY") expect_equal(out$row$gov_name, "MIAMI CITY")
}) })
@@ -241,7 +245,7 @@ test_that(".resolve_basket_row resolves with type override on ambiguous case", {
) )
expect_equal(out$status, "resolved") expect_equal(out$status, "resolved")
expect_equal(out$match_method, "substring") expect_equal(out$match_method, "substring")
expect_equal(out$row$canonical_govid, "052037010") expect_equal(out$row$canonical_govid, "062073207598")
}) })
# ---- basket mode public surface ---- # ---- basket mode public surface ----
@@ -254,7 +258,7 @@ test_that("cog_gov_search basket mode resolves clean inputs in input order", {
) )
expect_s3_class(basket, "tbl_df") expect_s3_class(basket, "tbl_df")
expect_equal(nrow(basket), 3L) expect_equal(nrow(basket), 3L)
expect_equal(basket$canonical_govid, c("101006006", "052037010", "442227001")) expect_equal(basket$canonical_govid, c("121011212191", "062073207598", "482453176394"))
expect_equal(basket$gov_name, c("BROWARD COUNTY", "SAN DIEGO CITY", "AUSTIN CITY")) expect_equal(basket$gov_name, c("BROWARD COUNTY", "SAN DIEGO CITY", "AUSTIN CITY"))
}) })
@@ -284,7 +288,7 @@ test_that("cog_gov_search basket mode skips ambiguous and no_match rows", {
)) ))
# Broward resolves; San Diego ambiguous; Notarealplace no_match. # Broward resolves; San Diego ambiguous; Notarealplace no_match.
expect_equal(nrow(basket), 1L) expect_equal(nrow(basket), 1L)
expect_equal(basket$canonical_govid, "101006006") expect_equal(basket$canonical_govid, "121011212191")
res <- attr(basket, "resolution") res <- attr(basket, "resolution")
expect_equal(nrow(res), 3L) expect_equal(nrow(res), 3L)
expect_equal(res$status, c("resolved", "ambiguous", "no_match")) expect_equal(res$status, c("resolved", "ambiguous", "no_match"))
@@ -306,20 +310,24 @@ test_that("cog_gov_search basket mode recycles single state", {
state = "CA" state = "CA"
) )
expect_equal(nrow(basket), 2L) expect_equal(nrow(basket), 2L)
expect_equal(basket$canonical_govid, c("052037010", "052001009")) expect_equal(basket$canonical_govid, c("062073207598", "062001123093"))
}) })
test_that("cog_gov_search basket mode within-type largest_pop records candidates", { test_that("cog_gov_search basket mode within-type largest_pop records candidates", {
skip_if_no_corpus() skip_if_no_corpus()
# `type = "city"` for the Miami row pins the match set to govs_type = 2;
# under Phase P canonical naming MIAMI-DADE COUNTY also contains "Miami"
# and would otherwise make this an ambiguous (cross-type) match.
basket <- suppressMessages(cog_gov_search( basket <- suppressMessages(cog_gov_search(
name = c("Miami", "OAKLAND CITY"), name = c("Miami", "OAKLAND CITY"),
state = c("FL", "CA") state = c("FL", "CA"),
type = c("city", NA)
)) ))
expect_equal(nrow(basket), 2L) expect_equal(nrow(basket), 2L)
res <- attr(basket, "resolution") res <- attr(basket, "resolution")
miami_row <- res[res$query_name == "Miami", ] miami_row <- res[res$query_name == "Miami", ]
expect_equal(miami_row$status, "largest_pop") expect_equal(miami_row$status, "largest_pop")
expect_equal(miami_row$canonical_govid, "102013013") expect_equal(miami_row$canonical_govid, "122086194757")
expect_gte(miami_row$n_candidates, 2L) expect_gte(miami_row$n_candidates, 2L)
expect_gte(nrow(miami_row$candidates[[1]]), 2L) expect_gte(nrow(miami_row$candidates[[1]]), 2L)
}) })
@@ -396,7 +404,7 @@ test_that("cog_gov_search basket mode skips per-row excluded type without aborti
)) ))
# Broward should resolve; the special_district row should be no_match. # Broward should resolve; the special_district row should be no_match.
expect_equal(nrow(basket), 1L) expect_equal(nrow(basket), 1L)
expect_equal(basket$canonical_govid, "101006006") expect_equal(basket$canonical_govid, "121011212191")
res <- attr(basket, "resolution") res <- attr(basket, "resolution")
expect_equal(res$status, c("resolved", "no_match")) expect_equal(res$status, c("resolved", "no_match"))
# query_type should record what the user passed for the excluded-type row # query_type should record what the user passed for the excluded-type row
+19 -18
View File
@@ -1,12 +1,12 @@
test_that("cog_spending returns expected shape for Broward Corrections 2020", { test_that("cog_spending returns expected shape for Broward Corrections 2020", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_spending("101006006", years = 2020L, category = "Corrections") r <- cog_spending("121011212191", years = 2020L, category = "Corrections")
expect_s3_class(r, "tbl_df") expect_s3_class(r, "tbl_df")
expected_cols <- c("year", "canonical_govid", "gov_name", "spend_subtype", expected_cols <- c("year", "canonical_govid", "gov_name", "spend_subtype",
"category", "amt_nominal", "codes_included", "category", "amt_nominal", "codes_included",
"aggregate_fallback", "notes") "aggregate_fallback", "notes")
expect_true(all(expected_cols %in% names(r))) expect_true(all(expected_cols %in% names(r)))
expect_equal(unique(r$canonical_govid), "101006006") expect_equal(unique(r$canonical_govid), "121011212191")
expect_equal(unique(r$year), 2020L) expect_equal(unique(r$year), 2020L)
expect_equal(unique(r$category), "Corrections") expect_equal(unique(r$category), "Corrections")
expect_true(all(r$spend_subtype %in% c("operations", "capital"))) expect_true(all(r$spend_subtype %in% c("operations", "capital")))
@@ -15,7 +15,7 @@ test_that("cog_spending returns expected shape for Broward Corrections 2020", {
test_that("cog_spending vectorised years + categories", { test_that("cog_spending vectorised years + categories", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_spending("101006006", 2019:2020, r <- cog_spending("121011212191", 2019:2020,
category = c("Corrections", "Police")) category = c("Corrections", "Police"))
expect_true(all(r$year %in% 2019:2020)) expect_true(all(r$year %in% 2019:2020))
expect_true(all(r$category %in% c("Corrections", "Police"))) expect_true(all(r$category %in% c("Corrections", "Police")))
@@ -24,7 +24,7 @@ test_that("cog_spending vectorised years + categories", {
test_that("cog_spending with per_capita adds per-capita nominal column", { test_that("cog_spending with per_capita adds per-capita nominal column", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_spending("101006006", 2020L, "Corrections", per_capita = TRUE) r <- cog_spending("121011212191", 2020L, "Corrections", per_capita = TRUE)
expect_true("amt_per_capita_nominal" %in% names(r)) expect_true("amt_per_capita_nominal" %in% names(r))
expect_false("amt_real" %in% names(r)) expect_false("amt_real" %in% names(r))
expect_false("amt_per_capita_real" %in% names(r)) expect_false("amt_per_capita_real" %in% names(r))
@@ -34,7 +34,7 @@ test_that("cog_spending with per_capita adds per-capita nominal column", {
test_that("cog_spending with adjust_to_year adds real column", { test_that("cog_spending with adjust_to_year adds real column", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_spending("101006006", 2019:2020, "Corrections", r <- cog_spending("121011212191", 2019:2020, "Corrections",
adjust_to_year = 2022L) adjust_to_year = 2022L)
expect_true("amt_real" %in% names(r)) expect_true("amt_real" %in% names(r))
r2019 <- dplyr::filter(r, year == 2019L) r2019 <- dplyr::filter(r, year == 2019L)
@@ -43,7 +43,7 @@ test_that("cog_spending with adjust_to_year adds real column", {
test_that("cog_spending with per_capita + adjust_to_year adds all columns", { test_that("cog_spending with per_capita + adjust_to_year adds all columns", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_spending("101006006", 2020L, "Corrections", r <- cog_spending("121011212191", 2020L, "Corrections",
per_capita = TRUE, adjust_to_year = 2022L) per_capita = TRUE, adjust_to_year = 2022L)
expect_true(all(c("amt_nominal", "amt_real", expect_true(all(c("amt_nominal", "amt_real",
"amt_per_capita_nominal", "amt_per_capita_real") %in% "amt_per_capita_nominal", "amt_per_capita_real") %in%
@@ -68,16 +68,16 @@ test_that("cog_spending for unknown govid returns empty tibble + informs", {
test_that("cog_spending records found + missing govids in provenance", { test_that("cog_spending records found + missing govids in provenance", {
skip_if_no_corpus() skip_if_no_corpus()
suppressMessages( suppressMessages(
r <- cog_spending(c("101006006", "XXXINVALID"), 2020L, "Corrections") r <- cog_spending(c("121011212191", "XXXINVALID"), 2020L, "Corrections")
) )
prov <- attr(r, "provenance") prov <- attr(r, "provenance")
expect_equal(sort(prov$scope$govids_found), "101006006") expect_equal(sort(prov$scope$govids_found), "121011212191")
expect_equal(sort(prov$scope$govids_missing), "XXXINVALID") expect_equal(sort(prov$scope$govids_missing), "XXXINVALID")
}) })
test_that("cog_spending result has provenance attribute matching schema", { test_that("cog_spending result has provenance attribute matching schema", {
skip_if_no_corpus() skip_if_no_corpus()
r <- cog_spending("101006006", 2020L, "Corrections") r <- cog_spending("121011212191", 2020L, "Corrections")
prov <- attr(r, "provenance") prov <- attr(r, "provenance")
expect_type(prov, "list") expect_type(prov, "list")
expect_equal(prov$verb, "cog_spending") expect_equal(prov$verb, "cog_spending")
@@ -93,7 +93,7 @@ test_that("cog_spending result has provenance attribute matching schema", {
test_that("cog_spending rejects invalid inputs", { test_that("cog_spending rejects invalid inputs", {
expect_error(cog_spending(list(), 2020L), "character|data frame") expect_error(cog_spending(list(), 2020L), "character|data frame")
expect_error(cog_spending("101006006", "2020"), "years") expect_error(cog_spending("121011212191", "2020"), "years")
}) })
test_that("cog_spending accepts a cog_gov_search result directly", { test_that("cog_spending accepts a cog_gov_search result directly", {
@@ -101,12 +101,12 @@ test_that("cog_spending accepts a cog_gov_search result directly", {
picks <- cog_gov_search("^BROWARD COUNTY$", state = "FL", type = "county") picks <- cog_gov_search("^BROWARD COUNTY$", state = "FL", type = "county")
expect_gt(nrow(picks), 0L) expect_gt(nrow(picks), 0L)
r <- cog_spending(picks, 2020L, "Corrections") r <- cog_spending(picks, 2020L, "Corrections")
expect_equal(unique(r$canonical_govid), "101006006") expect_equal(unique(r$canonical_govid), "121011212191")
}) })
test_that("cog_spending accepts a cog_find_peers result directly", { test_that("cog_spending accepts a cog_find_peers result directly", {
skip_if_no_corpus() skip_if_no_corpus()
peers <- cog_find_peers("101006006", max_peers = 3L) peers <- cog_find_peers("121011212191", max_peers = 3L)
r <- cog_spending(peers, 2020L, "Police") r <- cog_spending(peers, 2020L, "Police")
expect_setequal(unique(r$canonical_govid), expect_setequal(unique(r$canonical_govid),
sort(peers$canonical_govid)) sort(peers$canonical_govid))
@@ -132,7 +132,7 @@ test_that("cog_spending accepts a basket-mode cog_gov_search result", {
test_that("per_capita denominator is the per-year F-33 population", { test_that("per_capita denominator is the per-year F-33 population", {
skip_if_no_corpus() skip_if_no_corpus()
with_fixture_corpus({ with_fixture_corpus({
r <- cog_spending("101006006", years = 2019:2020, r <- cog_spending("121011212191", years = 2019:2020,
category = "Police", per_capita = TRUE) category = "Police", per_capita = TRUE)
r_ops <- r[r$spend_subtype == "operations", ] r_ops <- r[r$spend_subtype == "operations", ]
# Implied denominator from amt_nominal / amt_per_capita_nominal # Implied denominator from amt_nominal / amt_per_capita_nominal
@@ -140,9 +140,10 @@ test_that("per_capita denominator is the per-year F-33 population", {
names(implied_pop) <- r_ops$year names(implied_pop) <- r_ops$year
# Use absolute tolerance: within 1 person of per-year F-33 values. # Use absolute tolerance: within 1 person of per-year F-33 values.
# Hardcoded values are Broward County's per-year Census F-33 population # Hardcoded values are Broward County's per-year Census F-33 population
# from the bundled fixture (regenerated 2026-04-29 against cog_pipeline # from the bundled fixture (regenerated 2026-07-11 against cog_pipeline
# aad34c6 + bd3e744). 1,940,907 is the static ACS 2018-2022 5-year value # publish tree, pipeline_commit 1a00925, Phase P schema_version 4).
# the legacy implementation would use; we assert it is NOT what we get. # 1,940,907 is the static ACS 2018-2022 5-year value the legacy
# implementation would use; we assert it is NOT what we get.
expect_true(abs(implied_pop[["2019"]] - 1935878) < 1) expect_true(abs(implied_pop[["2019"]] - 1935878) < 1)
expect_true(abs(implied_pop[["2020"]] - 1952778) < 1) expect_true(abs(implied_pop[["2020"]] - 1952778) < 1)
expect_false(all(abs(implied_pop - 1940907) < 1)) expect_false(all(abs(implied_pop - 1940907) < 1))
@@ -152,7 +153,7 @@ test_that("per_capita denominator is the per-year F-33 population", {
test_that("pop_source = 'census_f33' does not produce unavailable-pop note", { test_that("pop_source = 'census_f33' does not produce unavailable-pop note", {
skip_if_no_corpus() skip_if_no_corpus()
with_fixture_corpus({ with_fixture_corpus({
r <- cog_spending("101006006", years = 2019L, r <- cog_spending("121011212191", years = 2019L,
category = "Police", per_capita = TRUE) category = "Police", per_capita = TRUE)
expect_true(all(r$pop_source == "census_f33")) expect_true(all(r$pop_source == "census_f33"))
expect_true(all(is.na(r$notes) | r$notes == "" | expect_true(all(is.na(r$notes) | r$notes == "" |
@@ -177,7 +178,7 @@ test_that("aggregate fallback + unavailable pop produce concatenated notes", {
test_that("provenance records per-year denominator metadata", { test_that("provenance records per-year denominator metadata", {
skip_if_no_corpus() skip_if_no_corpus()
with_fixture_corpus({ with_fixture_corpus({
r <- cog_spending("101006006", years = 2019:2020, r <- cog_spending("121011212191", years = 2019:2020,
category = "Police", per_capita = TRUE) category = "Police", per_capita = TRUE)
pc <- attr(r, "provenance")$transformations$per_capita pc <- attr(r, "provenance")$transformations$per_capita
expect_true(pc$applied) expect_true(pc$applied)
+5 -4
View File
@@ -66,15 +66,16 @@ test_that("gov_population_yearly exposes one row per (year, canonical_govid)", {
con, con,
"SELECT year, canonical_govid, population, popyear "SELECT year, canonical_govid, population, popyear
FROM gov_population_yearly FROM gov_population_yearly
WHERE canonical_govid = '101006006' WHERE canonical_govid = '121011212191'
ORDER BY year" ORDER BY year"
) )
expect_setequal(df$year, c(2019L, 2020L)) expect_setequal(df$year, c(2019L, 2020L))
expect_equal(nrow(df), 2L) expect_equal(nrow(df), 2L)
expect_true(all(!is.na(df$population))) expect_true(all(!is.na(df$population)))
# Hardcoded values are from the bundled fixture (regenerated 2026-04-29 # Hardcoded values are from the bundled fixture (regenerated 2026-07-11
# against cog_pipeline aad34c6 + bd3e744). Update if the fixture is # against cog_pipeline publish tree, pipeline_commit 1a00925, Phase P
# rebuilt against a different source vintage. # schema_version 4). Update if the fixture is rebuilt against a
# different source vintage.
expect_equal(df$population[df$year == 2019L], 1935878L) expect_equal(df$population[df$year == 2019L], 1935878L)
expect_equal(df$population[df$year == 2020L], 1952778L) expect_equal(df$population[df$year == 2020L], 1952778L)
# Uniqueness on (year, canonical_govid) across the whole view. # Uniqueness on (year, canonical_govid) across the whole view.
+2 -2
View File
@@ -48,8 +48,8 @@ Census sometimes uses a population estimate from one year prior to the fiscal ye
```r ```r
years <- 2010:2023 years <- 2010:2023
out <- purrr::map_dfr(years, function(y) { out <- purrr::map_dfr(years, function(y) {
peers <- cog_find_peers("231082082", year = y, max_peers = 10L) peers <- cog_find_peers("261163166615", year = y, max_peers = 10L)
cog_peer_compare("231082082", peers, cog_peer_compare("261163166615", peers,
category = "Police", years = y, category = "Police", years = y,
per_capita = TRUE) per_capita = TRUE)
}) })