The three kodor/fix issues, taken over after a day with no branch, PR or comment on any of them. Batched because each is single-file with a committed acceptance test, and two share documentation surfaces. #16 (F-025) -- cog_gov_search() utility mode interpolated `name` straight into regexp_matches() unescaped, while basket mode in the same file already routed it through .escape_regex() with the comment "so `name` is treated as a literal substring". Two failure modes, both HTTP 200 through the API: a government could not be found by its own complete name when that name contains a metacharacter (FREDONIA (BRISCOE) CITY returned nothing), and a bare "." matched all 608 Wisconsin cities. Malformed pattern text reached the engine as an error, which cog-api surfaced as a 500 -- reachable by typing a real name one character at a time ("Athens-Clarke County (bal"). Utility mode now calls the escaper that already existed. Roxygen updated: utility mode is documented as a literal case-insensitive substring match, and the basket-mode "substring fallback" step no longer describes itself as a regex either. BEHAVIOUR CHANGE worth flagging: anchored exact-match searches stop working, because there is no regex left to anchor. Two existing tests used "^BROWARD COUNTY$" and "^FLORIDA$" as their exact-match idiom; both now search for those characters literally. Updated to the bare names, which still resolve to exactly one row each once scoped by state/type (verified, not assumed). There is no exact-match option in utility mode any more -- noted on the issue, since that is a real if small capability loss. #15 (F-004) -- the raw Census files report thousands of dollars; this package multiplies by 1000 and returns full US dollars. Correct, and already stated in ?cog_spending / ?cog_revenue @return, in provenance, and in cog-api's data-dictionary. Absent from every surface a reader meets FIRST. Added to README.md as its own section and to both vignettes' openings. The dangerous one is cog_explorer/CLAUDE.md, which states the opposite rule ("All raw `amt` values are in $1,000s") without scoping it to the raw column -- a reader applying that to amt_nominal overstates by 1000x and gets a plausible-looking number rather than an obvious error. Fixed there too; that directory has no git remote, so it rides in no PR and is left uncommitted for the owner. #14 (F-021) -- .peer_summary_rows() computes stats::quantile() separately inside each (year, spend_subtype, category) cell, so a summary_p50 row is "the median peer's value in that one category", never "the value of the median peer's total" -- the median peer for Police and for Fire are usually different governments. Summing them across categories misstated a total-spending band by -32.7% to +251.0% across 24 years, with a sign flip at FY2012. The verb is right and its documented use (facet by role AND category) is unaffected, so the fix is @return prose plus a worked snippet showing the correct computation: sum each peer's own categories first, then take the quantile of those per-government totals. This is the R-side counterpart of cog-api#9, fixed on the API surface earlier today; the wording is deliberately consistent across the two. Note the phrase "not additive" has to stay on one roxygen source line -- the test greps the generated Rd, where a line wrap turns it into "not additive" and stops matching. Cost one red run to find. man/ regenerated with roxygen 8.0.0 against a repo built with 7.3.3, so cog_spending.Rd and DESCRIPTION were reverted -- their entire diff was version churn (reindentation, RoxygenNote -> Config/roxygen2/version) with no content change. The two Rd files kept carry only the edits above. Suite: 629 pass / 0 fail / 3 skip (was 606/0/6). The three remaining skips are #11, #12 and #13.
142 lines
5.3 KiB
R
142 lines
5.3 KiB
R
test_that("cog_geographic_rollup aggregates state + county + city layers", {
|
|
skip_if_no_corpus()
|
|
r <- cog_geographic_rollup(
|
|
govids = list(
|
|
state = "120000226351", # Florida state govt
|
|
county = "121011212191", # Broward County
|
|
city = "122011161585" # Fort Lauderdale City
|
|
),
|
|
category = "Police",
|
|
years = 2019:2020
|
|
)
|
|
expect_s3_class(r, "tbl_df")
|
|
expected_cols <- c("year", "layer", "canonical_govid", "gov_name",
|
|
"spend_subtype", "category", "amt_nominal",
|
|
"codes_included", "aggregate_fallback",
|
|
"scope_note", "notes")
|
|
expect_true(all(expected_cols %in% names(r)))
|
|
expect_setequal(unique(r$layer), c("state", "county", "city"))
|
|
expect_true(all(r$category == "Police"))
|
|
expect_true(all(r$year %in% 2019:2020))
|
|
})
|
|
|
|
test_that("cog_geographic_rollup respects per_capita + adjust_to_year", {
|
|
skip_if_no_corpus()
|
|
r <- cog_geographic_rollup(
|
|
govids = list(county = "121011212191", city = "122011161585"),
|
|
category = "Police",
|
|
years = 2020L,
|
|
per_capita = TRUE,
|
|
adjust_to_year = 2022L
|
|
)
|
|
expect_true(all(c("amt_nominal", "amt_real",
|
|
"amt_per_capita_nominal", "amt_per_capita_real") %in%
|
|
names(r)))
|
|
# Each layer's per-capita uses its own population: city pop < county pop,
|
|
# so per_capita_nominal for city rows should differ meaningfully from county.
|
|
city_pc <- r$amt_per_capita_nominal[r$layer == "city"]
|
|
cty_pc <- r$amt_per_capita_nominal[r$layer == "county"]
|
|
expect_true(length(city_pc) > 0L)
|
|
expect_true(length(cty_pc) > 0L)
|
|
})
|
|
|
|
test_that("cog_geographic_rollup scope_notes describe each layer", {
|
|
skip_if_no_corpus()
|
|
r <- cog_geographic_rollup(
|
|
govids = list(state = "120000226351", county = "121011212191",
|
|
city = "122011161585"),
|
|
category = "Police", years = 2020L
|
|
)
|
|
state_notes <- unique(r$scope_note[r$layer == "state"])
|
|
expect_true(any(grepl("state total", state_notes)))
|
|
county_notes <- unique(r$scope_note[r$layer == "county"])
|
|
expect_true(any(grepl("county", county_notes)))
|
|
city_notes <- unique(r$scope_note[r$layer == "city"])
|
|
expect_true(any(grepl("city proper", city_notes)))
|
|
})
|
|
|
|
test_that("cog_geographic_rollup single-layer call works", {
|
|
skip_if_no_corpus()
|
|
r <- cog_geographic_rollup(
|
|
govids = list(county = c("121011212191")),
|
|
category = "Corrections",
|
|
years = 2020L
|
|
)
|
|
expect_true(all(r$layer == "county"))
|
|
expect_gt(nrow(r), 0L)
|
|
})
|
|
|
|
test_that("cog_geographic_rollup provenance reports the outer verb", {
|
|
skip_if_no_corpus()
|
|
r <- cog_geographic_rollup(
|
|
govids = list(state = "120000226351", county = "121011212191"),
|
|
category = "Police", years = 2020L
|
|
)
|
|
prov <- attr(r, "provenance")
|
|
expect_equal(prov$verb, "cog_geographic_rollup")
|
|
expect_setequal(prov$layers, c("state", "county"))
|
|
expect_true(grepl("cog_geographic_rollup", prov$call))
|
|
})
|
|
|
|
test_that("cog_geographic_rollup accepts data.frames per layer", {
|
|
skip_if_no_corpus()
|
|
# Unanchored: utility mode matches literally now, so "^...$" would be
|
|
# searched for as characters rather than read as anchors (uscogdata#16).
|
|
# Both still resolve to exactly one row once scoped by type/state.
|
|
fl_state <- cog_gov_search("FLORIDA", type = "state")
|
|
broward <- cog_gov_search("BROWARD COUNTY", state = "FL", type = "county")
|
|
r <- cog_geographic_rollup(
|
|
govids = list(state = fl_state, county = broward),
|
|
category = "Police", years = 2020L
|
|
)
|
|
expect_setequal(unique(r$layer), c("state", "county"))
|
|
expect_gt(nrow(r), 0L)
|
|
})
|
|
|
|
test_that("cog_geographic_rollup rejects invalid inputs", {
|
|
expect_error(cog_geographic_rollup(list(), "Police", 2020L), "length")
|
|
expect_error(cog_geographic_rollup(c("121011212191"), "Police", 2020L), "list")
|
|
expect_error(
|
|
cog_geographic_rollup(list(planet = "120000226351"), "Police", 2020L),
|
|
"state|county|city"
|
|
)
|
|
})
|
|
|
|
test_that("cog_geographic_rollup per-capita uses summed per-year populations", {
|
|
skip_if_no_corpus()
|
|
with_fixture_corpus({
|
|
r <- cog_geographic_rollup(
|
|
govids = list(state = "010000226085",
|
|
county = "121011212191"),
|
|
category = "Police",
|
|
years = 2019:2020,
|
|
per_capita = TRUE
|
|
)
|
|
state_ops <- r[r$layer == "state" & r$spend_subtype == "operations", ]
|
|
county_ops <- r[r$layer == "county" & r$spend_subtype == "operations", ]
|
|
state_implied <- state_ops$amt_nominal / state_ops$amt_per_capita_nominal
|
|
county_implied <- county_ops$amt_nominal /
|
|
county_ops$amt_per_capita_nominal
|
|
# Per-year, per-layer denominator is the layer's own per-year population
|
|
expect_equal(state_implied[state_ops$year == 2019], 4874747, tolerance = 1)
|
|
expect_equal(state_implied[state_ops$year == 2020], 4903185, tolerance = 1)
|
|
expect_equal(county_implied[county_ops$year == 2019], 1935878, tolerance = 1)
|
|
})
|
|
})
|
|
|
|
test_that("cog_geographic_rollup records included/excluded govids in provenance", {
|
|
skip_if_no_corpus()
|
|
with_fixture_corpus({
|
|
r <- cog_geographic_rollup(
|
|
govids = list(county = "121011212191"),
|
|
category = "Police",
|
|
years = 2019:2020,
|
|
per_capita = TRUE
|
|
)
|
|
prov <- attr(r, "provenance")
|
|
expect_true("rollup" %in% names(prov))
|
|
expect_true("121011212191" %in% prov$rollup$included_govids)
|
|
expect_true(is.character(prov$rollup$excluded_govids))
|
|
})
|
|
})
|