Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2e8383b098
|
||
|
|
d006dea6e4
|
||
|
|
ebac39e6de | ||
|
|
47dc08c4b0 | ||
|
|
1d553a788f
|
@@ -1,5 +1,25 @@
|
||||
# uscogdata 0.1.0 (development)
|
||||
|
||||
## Corpus-wide series breaks now reach users (`corpus_break_refs`)
|
||||
|
||||
* Four catalogued series breaks carry `fin_code = "ALL"` — caveats about the
|
||||
corpus as a whole rather than about one item code. `series_break_refs` is
|
||||
built by matching `fin_code` against the item codes in the result, and no
|
||||
row's `item_code` is ever the literal `"ALL"`, so **none of them could ever
|
||||
be surfaced**: `SB085` (dollar precision across the 1976/1977 boundary),
|
||||
`SB087` (imputation exclusion from FY2002), `SB194` (the dense → sparse
|
||||
representation change at FY2012) and `SB086` (the government id scheme
|
||||
change at FY2017).
|
||||
* Provenance gains `corpus_break_refs`, selected on the break-year window
|
||||
alone and disjoint from `series_break_refs` by construction, so a consumer
|
||||
can tell a whole-result caveat from a break in one series. `cog_explain()`
|
||||
prints them under their own "Corpus-wide caveats" heading. cog-api passes
|
||||
provenance through verbatim, so the field appears there without an API
|
||||
change.
|
||||
* `SB194` is the one that made this urgent: a query spanning FY2011 → FY2012
|
||||
crosses the boundary where an absent cell stops meaning "Census published
|
||||
`$0`" and starts meaning "not reported", and until now nothing said so.
|
||||
|
||||
## Bundled fixture regenerated against the sparsified corpus
|
||||
|
||||
* `inst/extdata/fixture_corpus/` now tracks the corpus published on
|
||||
|
||||
@@ -123,6 +123,14 @@ cog_explain <- function(result, format = c("print", "list")) {
|
||||
cli::cli_ul(.series_break_story_lines(prov$series_break_refs))
|
||||
}
|
||||
|
||||
# Kept in a section of its own: these qualify the whole result, so folding
|
||||
# them in with the per-code breaks above would invite reading them as a
|
||||
# caveat about one series.
|
||||
if (length(prov$corpus_break_refs) > 0L) {
|
||||
cli::cli_h2("Corpus-wide caveats")
|
||||
cli::cli_ul(.series_break_story_lines(prov$corpus_break_refs))
|
||||
}
|
||||
|
||||
cli::cli_h2("Transformations")
|
||||
uc <- prov$transformations$units_conversion
|
||||
if (isTRUE(uc$applied)) {
|
||||
|
||||
@@ -133,7 +133,9 @@ cog_find_peers <- function(target_govid,
|
||||
#' [cog_find_peers()] result or a character vector of `canonical_govid`) and
|
||||
#' appends peer-distribution summary rows (`summary_p25`, `summary_p50`,
|
||||
#' `summary_p75`) so the result can be faceted by `role` in a single ggplot
|
||||
#' call.
|
||||
#' call. Those summary rows are quantiles **within each category**, not
|
||||
#' quantiles of each peer's total — see the `@return` section before summing
|
||||
#' them.
|
||||
#'
|
||||
#' @param target_govid Character scalar.
|
||||
#' @param peers A tibble from [cog_find_peers()] or a character vector of
|
||||
@@ -155,6 +157,32 @@ cog_find_peers <- function(target_govid,
|
||||
#' `attr(peers, "cohort_year")`; `NA` when `peers` was a bare character
|
||||
#' vector). Provenance reports `verb = "cog_peer_compare"`, `peer_count`,
|
||||
#' `cohort_year`, and `cohort_govids`.
|
||||
#'
|
||||
#' **The `summary_*` rows are per-category quantiles: they are not additive.**
|
||||
#' Each one is computed **within each `(year, spend_subtype,
|
||||
#' category)` cell** across the peer set, so a `summary_p50` row is *the
|
||||
#' median peer's value in that one category*, not *the value of the median
|
||||
#' peer's total*. The median peer for Police and the median peer for Fire
|
||||
#' are usually different governments, so summing `summary_*` rows across
|
||||
#' categories does not give any peer's total and misstates the band it
|
||||
#' appears to describe — measured at −32.7% to +251.0% across 24 years on
|
||||
#' one cohort, with a sign flip at FY2012.
|
||||
#'
|
||||
#' Facet by `role` **and** `category` (the documented use, and what the
|
||||
#' rows are built for). For a genuine "median peer's total spending" line,
|
||||
#' sum each peer's own categories first and take the quantile of those
|
||||
#' per-government totals:
|
||||
#'
|
||||
#' ```r
|
||||
#' library(dplyr)
|
||||
#' cmp |>
|
||||
#' filter(role %in% c("target", "peer")) |>
|
||||
#' group_by(year, role, canonical_govid) |>
|
||||
#' summarise(total = sum(amt_per_capita_real, na.rm = TRUE), .groups = "drop") |>
|
||||
#' filter(role == "peer") |>
|
||||
#' group_by(year) |>
|
||||
#' summarise(p50 = quantile(total, 0.5, na.rm = TRUE))
|
||||
#' ```
|
||||
#' @export
|
||||
cog_peer_compare <- function(target_govid, peers, category, years,
|
||||
per_capita = TRUE, adjust_to_year = NULL,
|
||||
|
||||
+11
-1
@@ -37,11 +37,20 @@
|
||||
|
||||
schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L))
|
||||
con <- .uscogdata_env$con
|
||||
break_refs <- if (!is.null(con) && DBI::dbIsValid(con)) {
|
||||
have_con <- !is.null(con) && DBI::dbIsValid(con)
|
||||
break_refs <- if (have_con) {
|
||||
.build_series_break_refs(con, codes_observed, years, schema_version)
|
||||
} else {
|
||||
character(0)
|
||||
}
|
||||
# Corpus-wide caveats travel separately: they qualify the whole result
|
||||
# rather than one series, and they do not depend on codes_observed (see
|
||||
# .build_corpus_break_refs()).
|
||||
corpus_refs <- if (have_con) {
|
||||
.build_corpus_break_refs(con, years, schema_version)
|
||||
} else {
|
||||
character(0)
|
||||
}
|
||||
|
||||
list(
|
||||
verb = verb,
|
||||
@@ -116,6 +125,7 @@
|
||||
)
|
||||
),
|
||||
series_break_refs = break_refs,
|
||||
corpus_break_refs = corpus_refs,
|
||||
manifest = list(
|
||||
schema_version = as.integer(manifest$schema_version),
|
||||
pipeline_commit = manifest$pipeline_commit %||% NA_character_,
|
||||
|
||||
+19
-7
@@ -6,8 +6,11 @@
|
||||
#' the cross-vintage canonical-government registry. Operates in two modes:
|
||||
#'
|
||||
#' * **Utility mode** (single `name`, the original behavior): returns all
|
||||
#' rows whose `gov_name` matches the regex case-insensitively, sorted by
|
||||
#' `population_acs` descending. Useful for exploratory lookups.
|
||||
#' rows whose `gov_name` contains `name` as a **literal, case-insensitive
|
||||
#' substring**, sorted by `population_acs` descending. Useful for
|
||||
#' exploratory lookups. Regex metacharacters in `name` are escaped, so a
|
||||
#' government is findable by its own complete name even when that name
|
||||
#' contains parentheses or a period.
|
||||
#' * **Basket mode** (`length(name) > 1`): resolves each input row to a
|
||||
#' single canonical govid and returns a tibble in input order, suitable
|
||||
#' for piping straight into [cog_spending()] / [cog_revenue()] /
|
||||
@@ -19,7 +22,8 @@
|
||||
#' 1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`.
|
||||
#' 2. **Exact pass:** case-insensitive equality against `gov_name`.
|
||||
#' Single hit -> resolved. Multiple -> step 4.
|
||||
#' 3. **Substring fallback:** case-insensitive regex against `gov_name`.
|
||||
#' 3. **Substring fallback:** case-insensitive literal substring against
|
||||
#' `gov_name` (metacharacters escaped).
|
||||
#' Single hit -> resolved (`match_method = "substring"`). Zero hits ->
|
||||
#' `status = "no_match"`. Multiple hits -> step 4.
|
||||
#' 4. **Disambiguation:** if matches share one `govs_type`, pick the
|
||||
@@ -48,7 +52,7 @@
|
||||
#' [cog_spending()], [cog_revenue()].
|
||||
#' @examples
|
||||
#' \dontrun{
|
||||
#' # Utility mode — exploratory regex lookup
|
||||
#' # Utility mode — exploratory substring lookup
|
||||
#' cog_gov_search("broward", state = "FL")
|
||||
#'
|
||||
#' # Basket mode — resolve a known cohort
|
||||
@@ -98,9 +102,16 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) {
|
||||
if (!is.character(name) || length(name) != 1L) {
|
||||
cli::cli_abort("`name` must be a length-1 character string.")
|
||||
}
|
||||
# Escaped, so `name` is a literal case-insensitive substring -- the same
|
||||
# treatment basket mode has always given it. Interpolating it raw made a
|
||||
# government unfindable by its own name whenever that name contains a
|
||||
# metacharacter (FREDONIA (BRISCOE) CITY), turned a bare "." into a
|
||||
# match-everything wildcard, and let malformed pattern text reach the
|
||||
# engine as an error -- which cog-api surfaced as a 500, reachable by
|
||||
# typing a real name one character at a time (uscogdata#16, F-025).
|
||||
preds <- c(preds,
|
||||
sprintf("regexp_matches(gov_name, %s, 'i')",
|
||||
.sql_lit_chr(name)))
|
||||
.sql_lit_chr(.escape_regex(name))))
|
||||
}
|
||||
if (!is.null(state)) {
|
||||
st_fips <- .coerce_state_to_fips(state)
|
||||
@@ -136,8 +147,9 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) {
|
||||
#' @noRd
|
||||
.escape_regex <- function(x) {
|
||||
# Backslash-escape POSIX regex metacharacters so `name` is treated as a
|
||||
# literal substring in the DuckDB regexp_matches call (substring fallback
|
||||
# only; utility-mode intentionally preserves regex behavior).
|
||||
# literal substring in the DuckDB regexp_matches call. Used by BOTH modes:
|
||||
# utility mode used to interpolate raw, which was a defect rather than a
|
||||
# feature -- see the call site and uscogdata#16.
|
||||
gsub("([\\^$.|?*+(){}\\[\\]])", "\\\\\\1", x, perl = TRUE)
|
||||
}
|
||||
|
||||
|
||||
+33
-1
@@ -14,9 +14,41 @@
|
||||
sql <- sprintf(
|
||||
"SELECT DISTINCT break_id
|
||||
FROM series_breaks_pq
|
||||
WHERE fin_code IN (%s) AND break_year BETWEEN %d AND %d
|
||||
WHERE fin_code IN (%s) AND fin_code <> 'ALL'
|
||||
AND break_year BETWEEN %d AND %d
|
||||
ORDER BY break_id",
|
||||
.sql_lit_chr(codes_observed), min(as.integer(years)), max(as.integer(years))
|
||||
)
|
||||
DBI::dbGetQuery(con, sql)$break_id
|
||||
}
|
||||
|
||||
#' Corpus-wide caveats: catalogued breaks whose `fin_code` is the literal
|
||||
#' `"ALL"` rather than an item code. They qualify the whole result, so they
|
||||
#' cannot be matched the way `.build_series_break_refs()` matches -- no row's
|
||||
#' `item_code` is ever `"ALL"`, which is exactly why they reached no user
|
||||
#' before uscogdata#19. Selection is on the break_year window alone: which
|
||||
#' codes a result happens to contain is irrelevant to a caveat about the
|
||||
#' corpus.
|
||||
#'
|
||||
#' All four catalogued entries are *boundary* caveats (dollar precision
|
||||
#' across 1976/1977, imputation exclusion from 2002, the dense -> sparse
|
||||
#' representation change at 2012, the id scheme change at 2017), so the same
|
||||
#' `break_year BETWEEN min(years) AND max(years)` rule the code-specific
|
||||
#' path uses is the right one -- a request that never crosses the boundary
|
||||
#' is not affected by it.
|
||||
#'
|
||||
#' Returned separately from `series_break_refs` so a consumer can tell a
|
||||
#' whole-result caveat from a break in one series; the two are disjoint by
|
||||
#' construction.
|
||||
#' @noRd
|
||||
.build_corpus_break_refs <- function(con, years, schema_version) {
|
||||
if (schema_version < 5L || length(years) == 0L) return(character(0))
|
||||
sql <- sprintf(
|
||||
"SELECT DISTINCT break_id
|
||||
FROM series_breaks_pq
|
||||
WHERE fin_code = 'ALL' AND break_year BETWEEN %d AND %d
|
||||
ORDER BY break_id",
|
||||
min(as.integer(years)), max(as.integer(years))
|
||||
)
|
||||
DBI::dbGetQuery(con, sql)$break_id
|
||||
}
|
||||
|
||||
@@ -19,6 +19,27 @@ package implements.
|
||||
# pak::pkg_install("gitea.civilytics.org/Civilytics/uscogdata")
|
||||
```
|
||||
|
||||
## Amounts are in full US dollars
|
||||
|
||||
Every amount column this package returns — `amt_nominal`, `amt_real`,
|
||||
`amt_per_capita_nominal`, `amt_per_capita_real` — is in **full US dollars**.
|
||||
|
||||
The raw Census source files report **thousands of dollars**, and the corpus's
|
||||
own `amt` column preserves that. The verbs multiply by 1000 on the way out, so
|
||||
you never have to. The conversion is recorded in every result:
|
||||
|
||||
```r
|
||||
r <- cog_spending("552025209777", 2020L)
|
||||
attr(r, "provenance")$transformations$units_conversion
|
||||
#> $applied TRUE $source_unit "$1,000s (raw Census)" $target_unit "$USD" $multiplier 1000
|
||||
```
|
||||
|
||||
**Do not multiply again.** If you have read elsewhere that COG amounts are in
|
||||
`$1,000s` — true of the raw corpus, and of `cog_explorer`'s conventions doc —
|
||||
that rule does not apply to anything a `cog_*()` verb hands you. Applying it
|
||||
twice overstates every figure by 1000x, and the result looks plausible rather
|
||||
than obviously wrong.
|
||||
|
||||
## Configuration
|
||||
|
||||
- `USCOGDATA_URL` — corpus root URL (public Nextcloud share, trailing slash)
|
||||
|
||||
@@ -33,6 +33,11 @@
|
||||
"aggregate_fallback": { "type": ["object", "null"] },
|
||||
"transformations":{ "type": "object" },
|
||||
"series_break_refs": { "type": "array", "items": { "type": "string" } },
|
||||
"corpus_break_refs": {
|
||||
"type": "array",
|
||||
"items": { "type": "string" },
|
||||
"description": "Ids of catalogued series breaks whose fin_code is the literal 'ALL' -- caveats about the corpus as a whole (dollar precision across 1976/1977, imputation exclusion from 2002, the dense -> sparse representation change at 2012, the government id scheme change at 2017) rather than about one item code. Selected on the break_year window alone, so they do not depend on which codes a result contains. Disjoint from series_break_refs by construction: an entry qualifies the whole result, not one series."
|
||||
},
|
||||
"manifest": { "type": "object" },
|
||||
"sql_query": { "type": "string" }
|
||||
}
|
||||
|
||||
@@ -32,8 +32,11 @@ the cross-vintage canonical-government registry. Operates in two modes:
|
||||
}
|
||||
\details{
|
||||
* **Utility mode** (single `name`, the original behavior): returns all
|
||||
rows whose `gov_name` matches the regex case-insensitively, sorted by
|
||||
`population_acs` descending. Useful for exploratory lookups.
|
||||
rows whose `gov_name` contains `name` as a **literal, case-insensitive
|
||||
substring**, sorted by `population_acs` descending. Useful for
|
||||
exploratory lookups. Regex metacharacters in `name` are escaped, so a
|
||||
government is findable by its own complete name even when that name
|
||||
contains parentheses or a period.
|
||||
* **Basket mode** (`length(name) > 1`): resolves each input row to a
|
||||
single canonical govid and returns a tibble in input order, suitable
|
||||
for piping straight into [cog_spending()] / [cog_revenue()] /
|
||||
@@ -45,7 +48,8 @@ the cross-vintage canonical-government registry. Operates in two modes:
|
||||
1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`.
|
||||
2. **Exact pass:** case-insensitive equality against `gov_name`.
|
||||
Single hit -> resolved. Multiple -> step 4.
|
||||
3. **Substring fallback:** case-insensitive regex against `gov_name`.
|
||||
3. **Substring fallback:** case-insensitive literal substring against
|
||||
`gov_name` (metacharacters escaped).
|
||||
Single hit -> resolved (`match_method = "substring"`). Zero hits ->
|
||||
`status = "no_match"`. Multiple hits -> step 4.
|
||||
4. **Disambiguation:** if matches share one `govs_type`, pick the
|
||||
@@ -58,7 +62,7 @@ inputs (`ambiguous` / `no_match`) appear only in the sidecar.
|
||||
}
|
||||
\examples{
|
||||
\dontrun{
|
||||
# Utility mode — exploratory regex lookup
|
||||
# Utility mode — exploratory substring lookup
|
||||
cog_gov_search("broward", state = "FL")
|
||||
|
||||
# Basket mode — resolve a known cohort
|
||||
|
||||
+29
-1
@@ -43,11 +43,39 @@ Tibble matching [cog_spending()]'s columns, plus a `role`
|
||||
`attr(peers, "cohort_year")`; `NA` when `peers` was a bare character
|
||||
vector). Provenance reports `verb = "cog_peer_compare"`, `peer_count`,
|
||||
`cohort_year`, and `cohort_govids`.
|
||||
|
||||
**The `summary_*` rows are per-category quantiles: they are not additive.**
|
||||
Each one is computed **within each `(year, spend_subtype,
|
||||
category)` cell** across the peer set, so a `summary_p50` row is *the
|
||||
median peer's value in that one category*, not *the value of the median
|
||||
peer's total*. The median peer for Police and the median peer for Fire
|
||||
are usually different governments, so summing `summary_*` rows across
|
||||
categories does not give any peer's total and misstates the band it
|
||||
appears to describe — measured at −32.7% to +251.0% across 24 years on
|
||||
one cohort, with a sign flip at FY2012.
|
||||
|
||||
Facet by `role` **and** `category` (the documented use, and what the
|
||||
rows are built for). For a genuine "median peer's total spending" line,
|
||||
sum each peer's own categories first and take the quantile of those
|
||||
per-government totals:
|
||||
|
||||
```r
|
||||
library(dplyr)
|
||||
cmp |>
|
||||
filter(role %in% c("target", "peer")) |>
|
||||
group_by(year, role, canonical_govid) |>
|
||||
summarise(total = sum(amt_per_capita_real, na.rm = TRUE), .groups = "drop") |>
|
||||
filter(role == "peer") |>
|
||||
group_by(year) |>
|
||||
summarise(p50 = quantile(total, 0.5, na.rm = TRUE))
|
||||
```
|
||||
}
|
||||
\description{
|
||||
Pulls spending for the target plus a peer set (either a
|
||||
[cog_find_peers()] result or a character vector of `canonical_govid`) and
|
||||
appends peer-distribution summary rows (`summary_p25`, `summary_p50`,
|
||||
`summary_p75`) so the result can be faceted by `role` in a single ggplot
|
||||
call.
|
||||
call. Those summary rows are quantiles **within each category**, not
|
||||
quantiles of each peer's total — see the `@return` section before summing
|
||||
them.
|
||||
}
|
||||
|
||||
@@ -6,6 +6,33 @@ fixture_corpus_path <- function() {
|
||||
if (nzchar(p)) paste0(p, "/") else ""
|
||||
}
|
||||
|
||||
# Path to a file in the SOURCE tree (README.md, man/*.Rd, vignettes/*.Rmd),
|
||||
# or "" when it isn't there.
|
||||
#
|
||||
# Tests that assert on documentation content have to read the sources, and the
|
||||
# sources only exist when the suite runs from a checkout. Under R CMD check the
|
||||
# suite runs from the INSTALLED package, where man/ and vignettes/ are not
|
||||
# shipped and `../../README.md` does not resolve -- so those tests must skip
|
||||
# rather than error. CI runs testthat::test_local() from the checkout BEFORE
|
||||
# rcmdcheck, so the assertions are still enforced on every push; this only
|
||||
# stops them from failing a context that structurally cannot satisfy them.
|
||||
source_tree_path <- function(...) {
|
||||
p <- testthat::test_path("..", "..", ...)
|
||||
if (file.exists(p)) p else ""
|
||||
}
|
||||
|
||||
# Skip unless every named source file is present (see source_tree_path()).
|
||||
skip_if_no_source_tree <- function(...) {
|
||||
paths <- vapply(list(...), function(rel) do.call(source_tree_path, as.list(rel)),
|
||||
character(1))
|
||||
missing <- vapply(paths, function(p) !nzchar(p), logical(1))
|
||||
testthat::skip_if(
|
||||
any(missing),
|
||||
"package source tree not available (running against the installed package)"
|
||||
)
|
||||
invisible(paths)
|
||||
}
|
||||
|
||||
# Skip a test if no corpus is reachable (bundled fixture or explicit remote URL).
|
||||
skip_if_no_corpus <- function() {
|
||||
p <- fixture_corpus_path()
|
||||
|
||||
@@ -17,7 +17,16 @@
|
||||
# cog-api's llms.txt, which is silent on units).
|
||||
|
||||
test_that("returned amounts are documented as full US dollars where readers meet the package", {
|
||||
testthat::skip("Blocked on uscogdata#15 (finding F-004)")
|
||||
|
||||
# README and vignettes ship only in the source tree, not in the installed
|
||||
# package, so these assertions cannot run under R CMD check -- CI's earlier
|
||||
# testthat::test_local() step is what enforces them. See
|
||||
# skip_if_no_source_tree() in helper-fixture.R.
|
||||
docs <- skip_if_no_source_tree(
|
||||
"README.md",
|
||||
c("vignettes", "total-spending.Rmd"),
|
||||
c("vignettes", "population-denominators.Rmd")
|
||||
)
|
||||
|
||||
says_units <- function(path) {
|
||||
txt <- paste(readLines(path, warn = FALSE), collapse = " ")
|
||||
@@ -25,10 +34,7 @@ test_that("returned amounts are documented as full US dollars where readers meet
|
||||
grepl("\\$1,000s|thousands of dollars", txt, ignore.case = TRUE)
|
||||
}
|
||||
|
||||
expect_true(says_units(testthat::test_path("..", "..", "README.md")))
|
||||
expect_true(says_units(testthat::test_path("..", "..", "vignettes", "total-spending.Rmd")))
|
||||
expect_true(says_units(testthat::test_path("..", "..", "vignettes",
|
||||
"population-denominators.Rmd")))
|
||||
for (path in docs) expect_true(says_units(path))
|
||||
|
||||
# Pin the documented claim to the actual behaviour, so the two cannot drift.
|
||||
# The expected raw amount is read straight from the corpus's parquet
|
||||
|
||||
@@ -0,0 +1,94 @@
|
||||
# tests/testthat/test-corpus-breaks.R
|
||||
#
|
||||
# uscogdata#19. Four catalogued series breaks carry fin_code = "ALL" -- they
|
||||
# are caveats about the corpus itself rather than about one item code:
|
||||
#
|
||||
# SB085 1977 dollar precision across the 1976/1977 boundary
|
||||
# SB087 2002 imputation exclusion FY2002-2006
|
||||
# SB194 2012 dense -> sparse representation change
|
||||
# SB086 2017 government ID scheme change
|
||||
#
|
||||
# .build_series_break_refs() matches `fin_code IN (<codes in the result>)`,
|
||||
# and no row's item_code is ever the literal "ALL", so none of them could
|
||||
# ever reach a user. They now travel in their own provenance field,
|
||||
# `corpus_break_refs`, which keeps them distinguishable from the
|
||||
# code-specific `series_break_refs` (an ALL caveat qualifies the whole
|
||||
# result, not one series).
|
||||
|
||||
test_that("corpus_break_refs surfaces an ALL-scoped break the year range spans", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
# SB194 sits at FY2012 -- the dense/sparse boundary. A query spanning
|
||||
# 2011 -> 2012 straddles it, and this is the case cog_pipeline#64's
|
||||
# DoD 4 intended to reach users.
|
||||
r <- cog_spending("121011212191", 2011:2012, "Police")
|
||||
prov <- attr(r, "provenance")
|
||||
expect_true("SB194" %in% prov$corpus_break_refs)
|
||||
})
|
||||
})
|
||||
|
||||
test_that("corpus_break_refs stays empty when no ALL break falls in the range", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
# 2019-2020 spans no catalogued corpus-wide break.
|
||||
r <- cog_spending("121011212191", 2019:2020, "Police")
|
||||
expect_equal(attr(r, "provenance")$corpus_break_refs, character(0))
|
||||
})
|
||||
})
|
||||
|
||||
test_that("corpus_break_refs and series_break_refs stay disjoint", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
r <- cog_spending("121011212191", 2011:2012, "Police")
|
||||
prov <- attr(r, "provenance")
|
||||
expect_type(prov$series_break_refs, "character")
|
||||
expect_type(prov$corpus_break_refs, "character")
|
||||
# An ALL caveat must never masquerade as a break in a specific series.
|
||||
expect_length(intersect(prov$series_break_refs, prov$corpus_break_refs), 0L)
|
||||
expect_false("SB194" %in% prov$series_break_refs)
|
||||
})
|
||||
})
|
||||
|
||||
test_that(".build_corpus_break_refs matches on the break_year window alone", {
|
||||
skip_if_no_corpus()
|
||||
con <- cog_open()
|
||||
on.exit(cog_close())
|
||||
|
||||
# SB085's boundary is 1976/1977, outside the fixture's partitions -- the
|
||||
# series_breaks table is a full cross-vintage registry, so the matching
|
||||
# logic is testable there even though no long partition covers it.
|
||||
expect_true("SB085" %in% uscogdata:::.build_corpus_break_refs(
|
||||
con, years = 1975:1980, schema_version = 6L
|
||||
))
|
||||
# ... and does not fire for a range that misses it, unlike a filter keyed
|
||||
# on the era rather than the boundary.
|
||||
expect_false("SB085" %in% uscogdata:::.build_corpus_break_refs(
|
||||
con, years = 1978:1980, schema_version = 6L
|
||||
))
|
||||
|
||||
# Unlike code-specific refs, these do not depend on which codes a result
|
||||
# happens to contain -- that dependency is the whole defect.
|
||||
expect_setequal(
|
||||
uscogdata:::.build_corpus_break_refs(con, years = 2001:2003, schema_version = 6L),
|
||||
"SB087"
|
||||
)
|
||||
|
||||
# Gated on schema_version >= 5: series_breaks_pq is not registered below it.
|
||||
expect_equal(
|
||||
uscogdata:::.build_corpus_break_refs(con, years = 2011:2012, schema_version = 4L),
|
||||
character(0)
|
||||
)
|
||||
})
|
||||
|
||||
test_that("cog_explain() prints corpus-wide caveats under their own heading", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
r <- cog_spending("121011212191", 2011:2012, "Police")
|
||||
out <- paste(c(
|
||||
capture.output(cog_explain(r)),
|
||||
capture.output(cog_explain(r), type = "message")
|
||||
), collapse = "\n")
|
||||
expect_match(out, "Corpus-wide caveats", fixed = TRUE)
|
||||
expect_match(out, "SB194", fixed = TRUE)
|
||||
})
|
||||
})
|
||||
@@ -18,7 +18,6 @@
|
||||
# semantics, not a row the fix makes findable.
|
||||
|
||||
test_that("cog_gov_search() matches name literally, not as an unescaped regex", {
|
||||
testthat::skip("Blocked on uscogdata#16 (finding F-025)")
|
||||
|
||||
# -- correctness (1): a government must be findable by its own exact name ---
|
||||
# FREDONIA (BRISCOE) CITY is real; today the parentheses are read as regex
|
||||
|
||||
@@ -13,10 +13,16 @@
|
||||
# is unaffected, so the fix is documentation: one sentence in @return.
|
||||
|
||||
test_that("cog_peer_compare() documents that summary_* rows are per-category quantiles", {
|
||||
testthat::skip("Blocked on uscogdata#14 (finding F-021)")
|
||||
|
||||
rd <- paste(readLines(testthat::test_path("..", "..", "man", "cog_peer_compare.Rd"),
|
||||
warn = FALSE), collapse = " ")
|
||||
# man/ ships only in the source tree (the installed package carries a
|
||||
# compiled help database instead), so the prose assertions below cannot run
|
||||
# under R CMD check -- CI's earlier testthat::test_local() step enforces
|
||||
# them. The numeric pin further down needs only the corpus, but it lives in
|
||||
# the same test_that() as the sentence it protects, deliberately: they are
|
||||
# one claim, and splitting them would let the prose drift while a separate
|
||||
# test kept passing.
|
||||
rd_path <- skip_if_no_source_tree(c("man", "cog_peer_compare.Rd"))
|
||||
rd <- paste(readLines(rd_path, warn = FALSE), collapse = " ")
|
||||
|
||||
# The @return section must say the quantile is computed within each cell...
|
||||
expect_match(rd, "within each|per-category|per category", ignore.case = TRUE)
|
||||
|
||||
@@ -80,8 +80,11 @@ test_that("cog_geographic_rollup provenance reports the outer verb", {
|
||||
|
||||
test_that("cog_geographic_rollup accepts data.frames per layer", {
|
||||
skip_if_no_corpus()
|
||||
fl_state <- cog_gov_search("^FLORIDA$", type = "state")
|
||||
broward <- cog_gov_search("^BROWARD COUNTY$", state = "FL", type = "county")
|
||||
# Unanchored: utility mode matches literally now, so "^...$" would be
|
||||
# searched for as characters rather than read as anchors (uscogdata#16).
|
||||
# Both still resolve to exactly one row once scoped by type/state.
|
||||
fl_state <- cog_gov_search("FLORIDA", type = "state")
|
||||
broward <- cog_gov_search("BROWARD COUNTY", state = "FL", type = "county")
|
||||
r <- cog_geographic_rollup(
|
||||
govids = list(state = fl_state, county = broward),
|
||||
category = "Police", years = 2020L
|
||||
|
||||
@@ -98,7 +98,11 @@ test_that("cog_spending rejects invalid inputs", {
|
||||
|
||||
test_that("cog_spending accepts a cog_gov_search result directly", {
|
||||
skip_if_no_corpus()
|
||||
picks <- cog_gov_search("^BROWARD COUNTY$", state = "FL", type = "county")
|
||||
# Unanchored: utility mode matches `name` as a literal substring now, so
|
||||
# "^...$" would be searched for as those characters rather than read as
|
||||
# anchors (uscogdata#16). Scoped by state and type, the bare name still
|
||||
# resolves to exactly one row.
|
||||
picks <- cog_gov_search("BROWARD COUNTY", state = "FL", type = "county")
|
||||
expect_gt(nrow(picks), 0L)
|
||||
r <- cog_spending(picks, 2020L, "Corrections")
|
||||
expect_equal(unique(r$canonical_govid), "121011212191")
|
||||
@@ -321,11 +325,13 @@ test_that("provenance$series_break_refs is a populated-when-applicable character
|
||||
r <- cog_spending("121011212191", 2020L, "Corrections")
|
||||
refs <- attr(r, "provenance")$series_break_refs
|
||||
expect_type(refs, "character")
|
||||
# No catalogued series_breaks_pq row falls inside this fixture's
|
||||
# 2011/2012/2019/2020 window for the codes this query touches (E04/G04)
|
||||
# -- data-verified; the mechanism itself is what's under test here, via
|
||||
# a query-shaped unit test in test-views.R since the fixture has no
|
||||
# positive case to pin against.
|
||||
# No catalogued code-specific series_breaks_pq row falls inside this
|
||||
# fixture's 2011/2012/2019/2020 window for the codes this query touches
|
||||
# (E04/G04) -- data-verified; the mechanism itself is what's under test
|
||||
# here, via a query-shaped unit test in test-views.R since the fixture
|
||||
# has no positive case to pin against. Corpus-wide ("ALL") entries never
|
||||
# appear in this field by construction -- they travel in
|
||||
# corpus_break_refs; see test-corpus-breaks.R.
|
||||
expect_equal(refs, character(0))
|
||||
})
|
||||
})
|
||||
|
||||
@@ -177,11 +177,13 @@ test_that("inst/sql/24- and 25- IG views retain aggregates, COALESCE NULL harmon
|
||||
})
|
||||
|
||||
test_that(".build_series_break_refs matches fin_code + break_year window", {
|
||||
# No series_breaks_pq row falls inside the bundled fixture's 2011-2020
|
||||
# window (data-verified; see the "series_break_refs" test in
|
||||
# No CODE-SPECIFIC series_breaks_pq row falls inside the bundled fixture's
|
||||
# 2011-2020 window (data-verified; see the "series_break_refs" test in
|
||||
# test-spending.R), so this proves the matching logic itself against the
|
||||
# live view + a synthetic year window that DOES hit a cataloged break
|
||||
# (SB075, fin_code E62, break_year 2005).
|
||||
# (SB075, fin_code E62, break_year 2005). The corpus-wide entries are a
|
||||
# separate path with its own coverage -- SB194 does sit at 2012, inside
|
||||
# the fixture window; see test-corpus-breaks.R.
|
||||
skip_if_no_corpus()
|
||||
con <- cog_open()
|
||||
on.exit(cog_close())
|
||||
|
||||
@@ -13,6 +13,8 @@ knitr::opts_chunk$set(eval = FALSE, collapse = TRUE, comment = "#>")
|
||||
|
||||
# Why per-year population matters
|
||||
|
||||
A note on units first, since every figure below is a rate: the numerator is in **full US dollars**. The raw Census files report **thousands of dollars** and the corpus keeps them that way in its own `amt` column, but `cog_spending()` and `cog_revenue()` multiply by 1000 on the way out, so `amt_per_capita_nominal` is already dollars per person. Do not scale it again.
|
||||
|
||||
Per-capita finance numbers divide each year's spending or revenue by a population denominator. The choice of denominator is a research decision, not an implementation detail: a 24-year corpus paired with a single 5-year ACS estimate produces biased per-capita values whose magnitude scales with each government's population change.
|
||||
|
||||
`uscogdata` defaults to the **Census F-33 population value Census itself uses to compute its published per-capita tables.** That value is recorded on every COG row as `population`, with `popyear` indicating the vintage. For a city that grew from 200,000 to 300,000 between 2000 and 2023, this default reproduces the per-capita value Census published. A static ACS denominator would have understated 2000 per-capita by ~33%.
|
||||
|
||||
@@ -29,6 +29,13 @@ controls which of these a query answers. This vignette walks through both
|
||||
questions with code that actually runs against the package's bundled fixture
|
||||
corpus, then explains why the second question refuses `"total"` outright.
|
||||
|
||||
Before any of the numbers below: every amount column here — `amt_nominal`,
|
||||
`amt_real`, and their `amt_per_capita_*` counterparts — is in **full US
|
||||
dollars**. The raw Census files report **thousands of dollars** and the
|
||||
corpus preserves that in its own `amt` column, but the verbs multiply by 1000
|
||||
on the way out. So `amt_nominal = 1317000` means $1.317 million, not $1.317
|
||||
billion. Do not scale it again.
|
||||
|
||||
```{r}
|
||||
library(uscogdata)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user