diff --git a/R/peers.R b/R/peers.R index 292453e..1919e35 100644 --- a/R/peers.R +++ b/R/peers.R @@ -240,6 +240,23 @@ cog_find_peers <- function(target_govid, #' group_by(year) |> #' summarise(p50 = quantile(total, 0.5, na.rm = TRUE)) #' ``` +#' @section Reading `coverage`: +#' `provenance$coverage` reports `n_units_reporting` against +#' `n_units_expected` per year. **`n_units_reporting` is category-conditional: +#' it counts cohort members with rows for the category you asked for, not +#' cohort members collected that year.** A government that was surveyed and +#' genuinely spends nothing in that category is indistinguishable here from one +#' that was never surveyed. +#' +#' The ratio is therefore **not a response rate** and must not be used as one. +#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for +#' `category = "Police"`; the 174-city gap is overwhelmingly cities that +#' contract policing to the county sheriff, not non-response. +#' +#' The comparison that *is* valid is the same category across a census year +#' (ending in 2 or 7) and a sample year, where the real-zero component is +#' roughly constant and the difference reflects the survey cycle. `is_census_year` +#' marks which is which. #' @export cog_peer_compare <- function(target_govid, peers, category, years, per_capita = TRUE, adjust_to_year = NULL, diff --git a/R/rollup.R b/R/rollup.R index 834fc8b..8edac2d 100644 --- a/R/rollup.R +++ b/R/rollup.R @@ -60,6 +60,23 @@ #' `codes_included`, `aggregate_fallback`, `scope_note`, `notes`. Carries a #' `provenance` attribute with `verb = "cog_geographic_rollup"`, `layers`, #' and `rollup$included_govids` / `rollup$excluded_govids`. +#' @section Reading `coverage`: +#' `provenance$coverage` reports `n_units_reporting` against +#' `n_units_expected` per year. **`n_units_reporting` is category-conditional: +#' it counts governments with rows for the category you asked for, not +#' governments collected that year.** A government that was surveyed and +#' genuinely spends nothing in that category is indistinguishable here from one +#' that was never surveyed. +#' +#' The ratio is therefore **not a response rate** and must not be used as one. +#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for +#' `category = "Police"`; the 174-city gap is overwhelmingly cities that +#' contract policing to the county sheriff, not non-response. +#' +#' The comparison that *is* valid is the same category across a census year +#' (ending in 2 or 7) and a sample year, where the real-zero component is +#' roughly constant and the difference reflects the survey cycle. `is_census_year` +#' marks which is which. #' @export cog_geographic_rollup <- function(govids, category, years, per_capita = FALSE, adjust_to_year = NULL, diff --git a/man/cog_geographic_rollup.Rd b/man/cog_geographic_rollup.Rd index d157b5b..15245a6 100644 --- a/man/cog_geographic_rollup.Rd +++ b/man/cog_geographic_rollup.Rd @@ -84,3 +84,23 @@ the result. The dropped govids are recorded in (gov type 4) and school districts (gov type 5) from per-capita rollups by design — see `vignette('population-denominators')`. } +\section{Reading `coverage`}{ + +`provenance$coverage` reports `n_units_reporting` against +`n_units_expected` per year. **`n_units_reporting` is category-conditional: +it counts governments with rows for the category you asked for, not +governments collected that year.** A government that was surveyed and +genuinely spends nothing in that category is indistinguishable here from one +that was never surveyed. + +The ratio is therefore **not a response rate** and must not be used as one. +In FY2022 — a complete census year — Georgia reports 393 of 567 cities for +`category = "Police"`; the 174-city gap is overwhelmingly cities that +contract policing to the county sheriff, not non-response. + +The comparison that *is* valid is the same category across a census year +(ending in 2 or 7) and a sample year, where the real-zero component is +roughly constant and the difference reflects the survey cycle. `is_census_year` +marks which is which. +} + diff --git a/man/cog_peer_compare.Rd b/man/cog_peer_compare.Rd index bd8e21a..c918195 100644 --- a/man/cog_peer_compare.Rd +++ b/man/cog_peer_compare.Rd @@ -107,3 +107,23 @@ call. Those summary rows are quantiles **within each category**, not quantiles of each peer's total — see the `@return` section before summing them. } +\section{Reading `coverage`}{ + +`provenance$coverage` reports `n_units_reporting` against +`n_units_expected` per year. **`n_units_reporting` is category-conditional: +it counts cohort members with rows for the category you asked for, not +cohort members collected that year.** A government that was surveyed and +genuinely spends nothing in that category is indistinguishable here from one +that was never surveyed. + +The ratio is therefore **not a response rate** and must not be used as one. +In FY2022 — a complete census year — Georgia reports 393 of 567 cities for +`category = "Police"`; the 174-city gap is overwhelmingly cities that +contract policing to the county sheriff, not non-response. + +The comparison that *is* valid is the same category across a census year +(ending in 2 or 7) and a sample year, where the real-zero component is +roughly constant and the difference reflects the survey cycle. `is_census_year` +marks which is which. +} + diff --git a/tests/testthat/test-all-categories-rollup.R b/tests/testthat/test-all-categories-rollup.R index 5e43f6e..5b979fe 100644 --- a/tests/testthat/test-all-categories-rollup.R +++ b/tests/testthat/test-all-categories-rollup.R @@ -38,3 +38,21 @@ test_that('cog_geographic_rollup() still refuses expenditure_concept = "total" w expenditure_concept = "total") ) }) + +test_that("n_units_reporting is category-conditional, not a response rate", { + skip_if_no_corpus() + govs <- cog_gov_search(name = NULL, state = "WI", type = 2L) + ids <- list(city = govs$canonical_govid) + + police <- cog_geographic_rollup(ids, category = "Police", years = 2012L) + allcat <- cog_geographic_rollup(ids, category = "All Categories", years = 2012L) + + cov_police <- cog_explain(police, format = "list")$coverage + cov_all <- cog_explain(allcat, format = "list")$coverage + + # Same year, same requested govids, same collection -- yet a single category + # reports fewer units than the all-categories query. That gap is real zeros, + # not non-response, which is exactly why the ratio is not a response rate. + expect_lte(cov_police$n_units_reporting, cov_all$n_units_reporting) + expect_identical(cov_police$n_units_expected, cov_all$n_units_expected) +})