fix: literal name search, units docs, peer-summary semantics (#16, #15, #14) #22

Merged
jared merged 2 commits from fix/kodor-batch-14-15-16 into main 2026-07-30 11:37:08 -04:00
13 changed files with 172 additions and 25 deletions
+29 -1
View File
@@ -133,7 +133,9 @@ cog_find_peers <- function(target_govid,
#' [cog_find_peers()] result or a character vector of `canonical_govid`) and #' [cog_find_peers()] result or a character vector of `canonical_govid`) and
#' appends peer-distribution summary rows (`summary_p25`, `summary_p50`, #' appends peer-distribution summary rows (`summary_p25`, `summary_p50`,
#' `summary_p75`) so the result can be faceted by `role` in a single ggplot #' `summary_p75`) so the result can be faceted by `role` in a single ggplot
#' call. #' call. Those summary rows are quantiles **within each category**, not
#' quantiles of each peer's total — see the `@return` section before summing
#' them.
#' #'
#' @param target_govid Character scalar. #' @param target_govid Character scalar.
#' @param peers A tibble from [cog_find_peers()] or a character vector of #' @param peers A tibble from [cog_find_peers()] or a character vector of
@@ -155,6 +157,32 @@ cog_find_peers <- function(target_govid,
#' `attr(peers, "cohort_year")`; `NA` when `peers` was a bare character #' `attr(peers, "cohort_year")`; `NA` when `peers` was a bare character
#' vector). Provenance reports `verb = "cog_peer_compare"`, `peer_count`, #' vector). Provenance reports `verb = "cog_peer_compare"`, `peer_count`,
#' `cohort_year`, and `cohort_govids`. #' `cohort_year`, and `cohort_govids`.
#'
#' **The `summary_*` rows are per-category quantiles: they are not additive.**
#' Each one is computed **within each `(year, spend_subtype,
#' category)` cell** across the peer set, so a `summary_p50` row is *the
#' median peer's value in that one category*, not *the value of the median
#' peer's total*. The median peer for Police and the median peer for Fire
#' are usually different governments, so summing `summary_*` rows across
#' categories does not give any peer's total and misstates the band it
#' appears to describe — measured at −32.7% to +251.0% across 24 years on
#' one cohort, with a sign flip at FY2012.
#'
#' Facet by `role` **and** `category` (the documented use, and what the
#' rows are built for). For a genuine "median peer's total spending" line,
#' sum each peer's own categories first and take the quantile of those
#' per-government totals:
#'
#' ```r
#' library(dplyr)
#' cmp |>
#' filter(role %in% c("target", "peer")) |>
#' group_by(year, role, canonical_govid) |>
#' summarise(total = sum(amt_per_capita_real, na.rm = TRUE), .groups = "drop") |>
#' filter(role == "peer") |>
#' group_by(year) |>
#' summarise(p50 = quantile(total, 0.5, na.rm = TRUE))
#' ```
#' @export #' @export
cog_peer_compare <- function(target_govid, peers, category, years, cog_peer_compare <- function(target_govid, peers, category, years,
per_capita = TRUE, adjust_to_year = NULL, per_capita = TRUE, adjust_to_year = NULL,
+19 -7
View File
@@ -6,8 +6,11 @@
#' the cross-vintage canonical-government registry. Operates in two modes: #' the cross-vintage canonical-government registry. Operates in two modes:
#' #'
#' * **Utility mode** (single `name`, the original behavior): returns all #' * **Utility mode** (single `name`, the original behavior): returns all
#' rows whose `gov_name` matches the regex case-insensitively, sorted by #' rows whose `gov_name` contains `name` as a **literal, case-insensitive
#' `population_acs` descending. Useful for exploratory lookups. #' substring**, sorted by `population_acs` descending. Useful for
#' exploratory lookups. Regex metacharacters in `name` are escaped, so a
#' government is findable by its own complete name even when that name
#' contains parentheses or a period.
#' * **Basket mode** (`length(name) > 1`): resolves each input row to a #' * **Basket mode** (`length(name) > 1`): resolves each input row to a
#' single canonical govid and returns a tibble in input order, suitable #' single canonical govid and returns a tibble in input order, suitable
#' for piping straight into [cog_spending()] / [cog_revenue()] / #' for piping straight into [cog_spending()] / [cog_revenue()] /
@@ -19,7 +22,8 @@
#' 1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`. #' 1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`.
#' 2. **Exact pass:** case-insensitive equality against `gov_name`. #' 2. **Exact pass:** case-insensitive equality against `gov_name`.
#' Single hit -> resolved. Multiple -> step 4. #' Single hit -> resolved. Multiple -> step 4.
#' 3. **Substring fallback:** case-insensitive regex against `gov_name`. #' 3. **Substring fallback:** case-insensitive literal substring against
#' `gov_name` (metacharacters escaped).
#' Single hit -> resolved (`match_method = "substring"`). Zero hits -> #' Single hit -> resolved (`match_method = "substring"`). Zero hits ->
#' `status = "no_match"`. Multiple hits -> step 4. #' `status = "no_match"`. Multiple hits -> step 4.
#' 4. **Disambiguation:** if matches share one `govs_type`, pick the #' 4. **Disambiguation:** if matches share one `govs_type`, pick the
@@ -48,7 +52,7 @@
#' [cog_spending()], [cog_revenue()]. #' [cog_spending()], [cog_revenue()].
#' @examples #' @examples
#' \dontrun{ #' \dontrun{
#' # Utility mode — exploratory regex lookup #' # Utility mode — exploratory substring lookup
#' cog_gov_search("broward", state = "FL") #' cog_gov_search("broward", state = "FL")
#' #'
#' # Basket mode — resolve a known cohort #' # Basket mode — resolve a known cohort
@@ -98,9 +102,16 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) {
if (!is.character(name) || length(name) != 1L) { if (!is.character(name) || length(name) != 1L) {
cli::cli_abort("`name` must be a length-1 character string.") cli::cli_abort("`name` must be a length-1 character string.")
} }
# Escaped, so `name` is a literal case-insensitive substring -- the same
# treatment basket mode has always given it. Interpolating it raw made a
# government unfindable by its own name whenever that name contains a
# metacharacter (FREDONIA (BRISCOE) CITY), turned a bare "." into a
# match-everything wildcard, and let malformed pattern text reach the
# engine as an error -- which cog-api surfaced as a 500, reachable by
# typing a real name one character at a time (uscogdata#16, F-025).
preds <- c(preds, preds <- c(preds,
sprintf("regexp_matches(gov_name, %s, 'i')", sprintf("regexp_matches(gov_name, %s, 'i')",
.sql_lit_chr(name))) .sql_lit_chr(.escape_regex(name))))
} }
if (!is.null(state)) { if (!is.null(state)) {
st_fips <- .coerce_state_to_fips(state) st_fips <- .coerce_state_to_fips(state)
@@ -136,8 +147,9 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) {
#' @noRd #' @noRd
.escape_regex <- function(x) { .escape_regex <- function(x) {
# Backslash-escape POSIX regex metacharacters so `name` is treated as a # Backslash-escape POSIX regex metacharacters so `name` is treated as a
# literal substring in the DuckDB regexp_matches call (substring fallback # literal substring in the DuckDB regexp_matches call. Used by BOTH modes:
# only; utility-mode intentionally preserves regex behavior). # utility mode used to interpolate raw, which was a defect rather than a
# feature -- see the call site and uscogdata#16.
gsub("([\\^$.|?*+(){}\\[\\]])", "\\\\\\1", x, perl = TRUE) gsub("([\\^$.|?*+(){}\\[\\]])", "\\\\\\1", x, perl = TRUE)
} }
+21
View File
@@ -19,6 +19,27 @@ package implements.
# pak::pkg_install("gitea.civilytics.org/Civilytics/uscogdata") # pak::pkg_install("gitea.civilytics.org/Civilytics/uscogdata")
``` ```
## Amounts are in full US dollars
Every amount column this package returns — `amt_nominal`, `amt_real`,
`amt_per_capita_nominal`, `amt_per_capita_real` — is in **full US dollars**.
The raw Census source files report **thousands of dollars**, and the corpus's
own `amt` column preserves that. The verbs multiply by 1000 on the way out, so
you never have to. The conversion is recorded in every result:
```r
r <- cog_spending("552025209777", 2020L)
attr(r, "provenance")$transformations$units_conversion
#> $applied TRUE $source_unit "$1,000s (raw Census)" $target_unit "$USD" $multiplier 1000
```
**Do not multiply again.** If you have read elsewhere that COG amounts are in
`$1,000s` — true of the raw corpus, and of `cog_explorer`'s conventions doc —
that rule does not apply to anything a `cog_*()` verb hands you. Applying it
twice overstates every figure by 1000x, and the result looks plausible rather
than obviously wrong.
## Configuration ## Configuration
- `USCOGDATA_URL` — corpus root URL (public Nextcloud share, trailing slash) - `USCOGDATA_URL` — corpus root URL (public Nextcloud share, trailing slash)
+8 -4
View File
@@ -32,8 +32,11 @@ the cross-vintage canonical-government registry. Operates in two modes:
} }
\details{ \details{
* **Utility mode** (single `name`, the original behavior): returns all * **Utility mode** (single `name`, the original behavior): returns all
rows whose `gov_name` matches the regex case-insensitively, sorted by rows whose `gov_name` contains `name` as a **literal, case-insensitive
`population_acs` descending. Useful for exploratory lookups. substring**, sorted by `population_acs` descending. Useful for
exploratory lookups. Regex metacharacters in `name` are escaped, so a
government is findable by its own complete name even when that name
contains parentheses or a period.
* **Basket mode** (`length(name) > 1`): resolves each input row to a * **Basket mode** (`length(name) > 1`): resolves each input row to a
single canonical govid and returns a tibble in input order, suitable single canonical govid and returns a tibble in input order, suitable
for piping straight into [cog_spending()] / [cog_revenue()] / for piping straight into [cog_spending()] / [cog_revenue()] /
@@ -45,7 +48,8 @@ the cross-vintage canonical-government registry. Operates in two modes:
1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`. 1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`.
2. **Exact pass:** case-insensitive equality against `gov_name`. 2. **Exact pass:** case-insensitive equality against `gov_name`.
Single hit -> resolved. Multiple -> step 4. Single hit -> resolved. Multiple -> step 4.
3. **Substring fallback:** case-insensitive regex against `gov_name`. 3. **Substring fallback:** case-insensitive literal substring against
`gov_name` (metacharacters escaped).
Single hit -> resolved (`match_method = "substring"`). Zero hits -> Single hit -> resolved (`match_method = "substring"`). Zero hits ->
`status = "no_match"`. Multiple hits -> step 4. `status = "no_match"`. Multiple hits -> step 4.
4. **Disambiguation:** if matches share one `govs_type`, pick the 4. **Disambiguation:** if matches share one `govs_type`, pick the
@@ -58,7 +62,7 @@ inputs (`ambiguous` / `no_match`) appear only in the sidecar.
} }
\examples{ \examples{
\dontrun{ \dontrun{
# Utility mode — exploratory regex lookup # Utility mode — exploratory substring lookup
cog_gov_search("broward", state = "FL") cog_gov_search("broward", state = "FL")
# Basket mode — resolve a known cohort # Basket mode — resolve a known cohort
+29 -1
View File
@@ -43,11 +43,39 @@ Tibble matching [cog_spending()]'s columns, plus a `role`
`attr(peers, "cohort_year")`; `NA` when `peers` was a bare character `attr(peers, "cohort_year")`; `NA` when `peers` was a bare character
vector). Provenance reports `verb = "cog_peer_compare"`, `peer_count`, vector). Provenance reports `verb = "cog_peer_compare"`, `peer_count`,
`cohort_year`, and `cohort_govids`. `cohort_year`, and `cohort_govids`.
**The `summary_*` rows are per-category quantiles: they are not additive.**
Each one is computed **within each `(year, spend_subtype,
category)` cell** across the peer set, so a `summary_p50` row is *the
median peer's value in that one category*, not *the value of the median
peer's total*. The median peer for Police and the median peer for Fire
are usually different governments, so summing `summary_*` rows across
categories does not give any peer's total and misstates the band it
appears to describe — measured at −32.7% to +251.0% across 24 years on
one cohort, with a sign flip at FY2012.
Facet by `role` **and** `category` (the documented use, and what the
rows are built for). For a genuine "median peer's total spending" line,
sum each peer's own categories first and take the quantile of those
per-government totals:
```r
library(dplyr)
cmp |>
filter(role %in% c("target", "peer")) |>
group_by(year, role, canonical_govid) |>
summarise(total = sum(amt_per_capita_real, na.rm = TRUE), .groups = "drop") |>
filter(role == "peer") |>
group_by(year) |>
summarise(p50 = quantile(total, 0.5, na.rm = TRUE))
```
} }
\description{ \description{
Pulls spending for the target plus a peer set (either a Pulls spending for the target plus a peer set (either a
[cog_find_peers()] result or a character vector of `canonical_govid`) and [cog_find_peers()] result or a character vector of `canonical_govid`) and
appends peer-distribution summary rows (`summary_p25`, `summary_p50`, appends peer-distribution summary rows (`summary_p25`, `summary_p50`,
`summary_p75`) so the result can be faceted by `role` in a single ggplot `summary_p75`) so the result can be faceted by `role` in a single ggplot
call. call. Those summary rows are quantiles **within each category**, not
quantiles of each peer's total — see the `@return` section before summing
them.
} }
+27
View File
@@ -6,6 +6,33 @@ fixture_corpus_path <- function() {
if (nzchar(p)) paste0(p, "/") else "" if (nzchar(p)) paste0(p, "/") else ""
} }
# Path to a file in the SOURCE tree (README.md, man/*.Rd, vignettes/*.Rmd),
# or "" when it isn't there.
#
# Tests that assert on documentation content have to read the sources, and the
# sources only exist when the suite runs from a checkout. Under R CMD check the
# suite runs from the INSTALLED package, where man/ and vignettes/ are not
# shipped and `../../README.md` does not resolve -- so those tests must skip
# rather than error. CI runs testthat::test_local() from the checkout BEFORE
# rcmdcheck, so the assertions are still enforced on every push; this only
# stops them from failing a context that structurally cannot satisfy them.
source_tree_path <- function(...) {
p <- testthat::test_path("..", "..", ...)
if (file.exists(p)) p else ""
}
# Skip unless every named source file is present (see source_tree_path()).
skip_if_no_source_tree <- function(...) {
paths <- vapply(list(...), function(rel) do.call(source_tree_path, as.list(rel)),
character(1))
missing <- vapply(paths, function(p) !nzchar(p), logical(1))
testthat::skip_if(
any(missing),
"package source tree not available (running against the installed package)"
)
invisible(paths)
}
# Skip a test if no corpus is reachable (bundled fixture or explicit remote URL). # Skip a test if no corpus is reachable (bundled fixture or explicit remote URL).
skip_if_no_corpus <- function() { skip_if_no_corpus <- function() {
p <- fixture_corpus_path() p <- fixture_corpus_path()
+11 -5
View File
@@ -17,7 +17,16 @@
# cog-api's llms.txt, which is silent on units). # cog-api's llms.txt, which is silent on units).
test_that("returned amounts are documented as full US dollars where readers meet the package", { test_that("returned amounts are documented as full US dollars where readers meet the package", {
testthat::skip("Blocked on uscogdata#15 (finding F-004)")
# README and vignettes ship only in the source tree, not in the installed
# package, so these assertions cannot run under R CMD check -- CI's earlier
# testthat::test_local() step is what enforces them. See
# skip_if_no_source_tree() in helper-fixture.R.
docs <- skip_if_no_source_tree(
"README.md",
c("vignettes", "total-spending.Rmd"),
c("vignettes", "population-denominators.Rmd")
)
says_units <- function(path) { says_units <- function(path) {
txt <- paste(readLines(path, warn = FALSE), collapse = " ") txt <- paste(readLines(path, warn = FALSE), collapse = " ")
@@ -25,10 +34,7 @@ test_that("returned amounts are documented as full US dollars where readers meet
grepl("\\$1,000s|thousands of dollars", txt, ignore.case = TRUE) grepl("\\$1,000s|thousands of dollars", txt, ignore.case = TRUE)
} }
expect_true(says_units(testthat::test_path("..", "..", "README.md"))) for (path in docs) expect_true(says_units(path))
expect_true(says_units(testthat::test_path("..", "..", "vignettes", "total-spending.Rmd")))
expect_true(says_units(testthat::test_path("..", "..", "vignettes",
"population-denominators.Rmd")))
# Pin the documented claim to the actual behaviour, so the two cannot drift. # Pin the documented claim to the actual behaviour, so the two cannot drift.
# The expected raw amount is read straight from the corpus's parquet # The expected raw amount is read straight from the corpus's parquet
@@ -18,7 +18,6 @@
# semantics, not a row the fix makes findable. # semantics, not a row the fix makes findable.
test_that("cog_gov_search() matches name literally, not as an unescaped regex", { test_that("cog_gov_search() matches name literally, not as an unescaped regex", {
testthat::skip("Blocked on uscogdata#16 (finding F-025)")
# -- correctness (1): a government must be findable by its own exact name --- # -- correctness (1): a government must be findable by its own exact name ---
# FREDONIA (BRISCOE) CITY is real; today the parentheses are read as regex # FREDONIA (BRISCOE) CITY is real; today the parentheses are read as regex
+9 -3
View File
@@ -13,10 +13,16 @@
# is unaffected, so the fix is documentation: one sentence in @return. # is unaffected, so the fix is documentation: one sentence in @return.
test_that("cog_peer_compare() documents that summary_* rows are per-category quantiles", { test_that("cog_peer_compare() documents that summary_* rows are per-category quantiles", {
testthat::skip("Blocked on uscogdata#14 (finding F-021)")
rd <- paste(readLines(testthat::test_path("..", "..", "man", "cog_peer_compare.Rd"), # man/ ships only in the source tree (the installed package carries a
warn = FALSE), collapse = " ") # compiled help database instead), so the prose assertions below cannot run
# under R CMD check -- CI's earlier testthat::test_local() step enforces
# them. The numeric pin further down needs only the corpus, but it lives in
# the same test_that() as the sentence it protects, deliberately: they are
# one claim, and splitting them would let the prose drift while a separate
# test kept passing.
rd_path <- skip_if_no_source_tree(c("man", "cog_peer_compare.Rd"))
rd <- paste(readLines(rd_path, warn = FALSE), collapse = " ")
# The @return section must say the quantile is computed within each cell... # The @return section must say the quantile is computed within each cell...
expect_match(rd, "within each|per-category|per category", ignore.case = TRUE) expect_match(rd, "within each|per-category|per category", ignore.case = TRUE)
+5 -2
View File
@@ -80,8 +80,11 @@ test_that("cog_geographic_rollup provenance reports the outer verb", {
test_that("cog_geographic_rollup accepts data.frames per layer", { test_that("cog_geographic_rollup accepts data.frames per layer", {
skip_if_no_corpus() skip_if_no_corpus()
fl_state <- cog_gov_search("^FLORIDA$", type = "state") # Unanchored: utility mode matches literally now, so "^...$" would be
broward <- cog_gov_search("^BROWARD COUNTY$", state = "FL", type = "county") # searched for as characters rather than read as anchors (uscogdata#16).
# Both still resolve to exactly one row once scoped by type/state.
fl_state <- cog_gov_search("FLORIDA", type = "state")
broward <- cog_gov_search("BROWARD COUNTY", state = "FL", type = "county")
r <- cog_geographic_rollup( r <- cog_geographic_rollup(
govids = list(state = fl_state, county = broward), govids = list(state = fl_state, county = broward),
category = "Police", years = 2020L category = "Police", years = 2020L
+5 -1
View File
@@ -98,7 +98,11 @@ test_that("cog_spending rejects invalid inputs", {
test_that("cog_spending accepts a cog_gov_search result directly", { test_that("cog_spending accepts a cog_gov_search result directly", {
skip_if_no_corpus() skip_if_no_corpus()
picks <- cog_gov_search("^BROWARD COUNTY$", state = "FL", type = "county") # Unanchored: utility mode matches `name` as a literal substring now, so
# "^...$" would be searched for as those characters rather than read as
# anchors (uscogdata#16). Scoped by state and type, the bare name still
# resolves to exactly one row.
picks <- cog_gov_search("BROWARD COUNTY", state = "FL", type = "county")
expect_gt(nrow(picks), 0L) expect_gt(nrow(picks), 0L)
r <- cog_spending(picks, 2020L, "Corrections") r <- cog_spending(picks, 2020L, "Corrections")
expect_equal(unique(r$canonical_govid), "121011212191") expect_equal(unique(r$canonical_govid), "121011212191")
+2
View File
@@ -13,6 +13,8 @@ knitr::opts_chunk$set(eval = FALSE, collapse = TRUE, comment = "#>")
# Why per-year population matters # Why per-year population matters
A note on units first, since every figure below is a rate: the numerator is in **full US dollars**. The raw Census files report **thousands of dollars** and the corpus keeps them that way in its own `amt` column, but `cog_spending()` and `cog_revenue()` multiply by 1000 on the way out, so `amt_per_capita_nominal` is already dollars per person. Do not scale it again.
Per-capita finance numbers divide each year's spending or revenue by a population denominator. The choice of denominator is a research decision, not an implementation detail: a 24-year corpus paired with a single 5-year ACS estimate produces biased per-capita values whose magnitude scales with each government's population change. Per-capita finance numbers divide each year's spending or revenue by a population denominator. The choice of denominator is a research decision, not an implementation detail: a 24-year corpus paired with a single 5-year ACS estimate produces biased per-capita values whose magnitude scales with each government's population change.
`uscogdata` defaults to the **Census F-33 population value Census itself uses to compute its published per-capita tables.** That value is recorded on every COG row as `population`, with `popyear` indicating the vintage. For a city that grew from 200,000 to 300,000 between 2000 and 2023, this default reproduces the per-capita value Census published. A static ACS denominator would have understated 2000 per-capita by ~33%. `uscogdata` defaults to the **Census F-33 population value Census itself uses to compute its published per-capita tables.** That value is recorded on every COG row as `population`, with `popyear` indicating the vintage. For a city that grew from 200,000 to 300,000 between 2000 and 2023, this default reproduces the per-capita value Census published. A static ACS denominator would have understated 2000 per-capita by ~33%.
+7
View File
@@ -29,6 +29,13 @@ controls which of these a query answers. This vignette walks through both
questions with code that actually runs against the package's bundled fixture questions with code that actually runs against the package's bundled fixture
corpus, then explains why the second question refuses `"total"` outright. corpus, then explains why the second question refuses `"total"` outright.
Before any of the numbers below: every amount column here — `amt_nominal`,
`amt_real`, and their `amt_per_capita_*` counterparts — is in **full US
dollars**. The raw Census files report **thousands of dollars** and the
corpus preserves that in its own `amt` column, but the verbs multiply by 1000
on the way out. So `amt_nominal = 1317000` means $1.317 million, not $1.317
billion. Do not scale it again.
```{r} ```{r}
library(uscogdata) library(uscogdata)