From b41d5ee2aa01f9efc7935ac1bb2d4f2f956a5fcc Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Wed, 9 Sep 2026 11:46:17 -0400 Subject: [PATCH] feat(coverage): n_units_collected separates sampling from real zeros (#36) provenance$coverage's n_units_reporting is category-conditional: it counts governments with rows for the SPECIFIC requested category, which conflates two different things -- a government never collected that year (sampling), and one collected but genuinely spending nothing in that category (a real zero). FY2012 Georgia Police is the motivating case from the issue: a complete census year reads as a 69% "response rate" because most of the gap is cities that contract policing to the county sheriff, not non-response. Adds a second counter, n_units_collected: how many of the caller's expected cohort appear in the corpus that year for ANY category. n_units_collected / n_units_expected is the true collection rate; n_units_reporting / n_units_collected is category participation among collected units. cog_geographic_rollup() and cog_peer_compare() both carry it; cog_explain() prints it alongside n_units_reporting. Two real bugs caught and fixed while finishing this (both against the already-written, previously-uncommitted draft): - .coverage_table()'s candidate list for the collection query was derived from the category-filtered result rows, not the caller's full expected cohort. A government with zero rows in the requested category across every requested year never appears in that result, so it was silently excluded from n_units_collected too -- collapsing the new counter back to the old, broken one for exactly the governments it exists to count. Fixed by threading an explicit `expected_ids` (all_govids / peer_govids) through instead. - The collection query hardcoded long_view = "spending_long_harmonized", which does not exist on a corpus with schema_version < 5 (R/basis.R resolves basis = "raw" there; R/views.R only registers the harmonized views on v5+). cog_geographic_rollup()/cog_peer_compare() would hard-error on a corpus vintage the package otherwise explicitly supports. Fixed by deriving long_view from the basis cog_spending() actually resolved (prov$basis) via the existing .select_long_view() helper, matching how every other basis-aware query in the package already does this. Also: cog_explain()'s general "complete census only in years ending in 2 or 7" footnote was gated on the OLD counter's absence, making it permanently unreachable now that both callers always supply the new one -- ungated it, since the explanation is orthogonal to which counter set is present. Dropped a dead conditional branch, fixed two stale roxygen blocks in R/peers.R/R/rollup.R still describing the old two-counter model, fixed the same staleness in README.md, and switched two `uscogdata:::` self-references to the package's own convention of calling internal helpers unqualified. 1101 tests pass (2 skipped live-corpus), including new direct regression tests for both bugs above (one exercising a government collected-but-absent from a category result, one running the full rollup/peer-compare path against a doctored schema_version 4 corpus). Reviewed by an independent code-reviewer pass (1 HIGH, 1 MEDIUM, 3 LOW -- all addressed above). Closes #36. Co-Authored-By: Claude Sonnet 5 --- R/coverage.R | 82 ++++++++++++++-- R/explain.R | 30 ++++-- R/peers.R | 47 +++++++--- R/rollup.R | 49 +++++++--- README.md | 13 ++- man/cog_geographic_rollup.Rd | 35 ++++--- man/cog_peer_compare.Rd | 35 ++++--- tests/testthat/test-coverage-disclosure.R | 108 ++++++++++++++++++++++ tests/testthat/test-explain.R | 14 +++ 9 files changed, 348 insertions(+), 65 deletions(-) diff --git a/R/coverage.R b/R/coverage.R index 3094f14..516bd25 100644 --- a/R/coverage.R +++ b/R/coverage.R @@ -84,22 +84,92 @@ #' `n_units_reporting = 0`, which is precisely the disclosure a silently #' missing year fails to make. #' -#' `n_units_reporting` describes the result the caller actually received, so -#' under `coverage = "consistent"` it reports the balanced count. `is_census_year` -#' is a statement about the SURVEY CALENDAR, never a claim of completeness: -#' FY1967 is a census year in which only 97 of Wisconsin's 608 cities report. -#' `n_units_reporting` is the number that tells the truth. +#' Three counters are returned, each answering a different question: +#' +#' * `n_units_expected` -- the universe the caller named (govids passed in, +#' or peers for cog_peer_compare). "How many governments did you ask +#' about?" +#' * `n_units_collected` -- how many of those appear in the corpus at all +#' that year, in ANY category. This is a statement about survey collection, +#' independent of what was asked for: "of the governments you named, how +#' many did Census actually collect data from this year?" It separates +#' sampling (not collected) from real zeros (collected but spends nothing +#' in your category). +#' * `n_units_reporting` -- how many of those appear with rows for the +#' SPECIFIC category you requested. This is always <= n_units_collected: +#' a government can be collected but have no rows for "Police" because it +#' contracts policing to the county sheriff, not because it wasn't +#' surveyed. +#' +#' `n_units_reporting` therefore conflates two very different things: a unit +#' that was not collected (sampling) and a unit that was collected but spends +#' nothing in that category. The ratio n_units_collected / n_units_expected is +#' the true collection rate; n_units_reporting / n_units_collected measures +#' category participation among collected units. +#' +#' `is_census_year` is a statement about the SURVEY CALENDAR, never a claim of +#' completeness: FY1967 is a census year in which only 97 of Wisconsin's 608 +#' cities report. The counters are what tell the truth. +#' +#' @param con Active DuckDB connection (used to look up n_units_collected). +#' @param long_view The verb's own long view, used for the collection query; +#' NULL skips the lookup and leaves n_units_collected as NA_integer_. +#' @param expected_ids The full EXPECTED cohort (govids the caller named), +#' used as the candidate list for the collection query. Required alongside +#' `con`/`long_view` for a correct count -- see the note below on why it +#' must not be derived from `result`/`rows`. `NULL`, or non-`NULL` but +#' empty after dropping `NA`/`""` entries, skips the lookup and leaves +#' n_units_collected as NA_integer_. #' @noRd .coverage_table <- function(result, years, n_expected, - id_col = "canonical_govid", rows = NULL) { + id_col = "canonical_govid", rows = NULL, + con = NULL, long_view = NULL, + expected_ids = NULL) { years <- sort(unique(as.integer(years))) src <- if (is.null(rows)) result else rows reporting <- vapply(years, function(y) { ids <- src[[id_col]][as.integer(src$year) == y] length(unique(ids[!is.na(ids)])) }, integer(1)) + + # n_units_collected: count EXPECTED cohort members present in the corpus + # for ANY category that year, not just the requested one. This separates + # sampling (not collected at all) from real zeros (collected but no rows + # for this category). Only computed when a connection, long_view, AND + # expected_ids are all provided; otherwise NA_integer_. + # + # The candidate list MUST be expected_ids, not derived from `result`/ + # `rows`: a government with zero rows in the requested category across + # EVERY requested year never appears in `result` at all, so deriving + # candidates from it would silently exclude exactly the "collected but + # real zero" governments this counter exists to count -- collapsing + # n_units_collected back to n_units_reporting for precisely the case #36 + # was filed over. + if (!is.null(con) && !is.null(long_view) && length(expected_ids) > 0L) { + cohort_chr <- .sql_lit_chr(unique(expected_ids[!is.na(expected_ids) & + nzchar(expected_ids)])) + years_lit <- paste(years, collapse = ",") + collected_q <- sprintf( + "SELECT year, COUNT(DISTINCT canonical_govid) AS n + FROM %s + WHERE canonical_govid IN (%s) + AND year IN (%s) + GROUP BY year", + long_view, cohort_chr, years_lit + ) + collected_df <- DBI::dbGetQuery(con, collected_q) + collected_map <- setNames(collected_df$n, as.integer(collected_df$year)) + collected <- vapply(years, function(y) { + val <- collected_map[as.character(y)] + if (is.na(val)) 0L else as.integer(val) + }, integer(1)) + } else { + collected <- rep(NA_integer_, length(years)) + } + tibble::tibble( year = years, + n_units_collected = collected, n_units_reporting = as.integer(reporting), n_units_expected = rep(as.integer(n_expected), length(years)), is_census_year = .is_census_year(years) diff --git a/R/explain.R b/R/explain.R index 417d35a..1505fd0 100644 --- a/R/explain.R +++ b/R/explain.R @@ -166,12 +166,30 @@ cog_explain <- function(result, format = c("print", "list")) { cli::cli_h2("Reporting coverage") cli::cli_text("Mode: {prov$coverage_mode %||% 'all'}") cov <- prov$coverage - cli::cli_ul(sprintf( - "%d: %d of %d units reporting (%.0f%%) -- %s year", - cov$year, cov$n_units_reporting, cov$n_units_expected, - 100 * cov$n_units_reporting / pmax(cov$n_units_expected, 1L), - ifelse(cov$is_census_year, "census", "sample") - )) + has_collected <- "n_units_collected" %in% names(cov) + if (has_collected) { + # Three counters: collected separates sampling from real zeros; + # reporting is category-conditional and never a response rate. + cli::cli_ul(sprintf( + "%d: %d of %d units collected, %d reporting in this category -- %s year", + cov$year, + cov$n_units_collected, + cov$n_units_expected, + cov$n_units_reporting, + ifelse(cov$is_census_year, "census", "sample") + )) + } else { + cli::cli_ul(sprintf( + "%d: %d of %d units reporting (%.0f%%) -- %s year", + cov$year, cov$n_units_reporting, cov$n_units_expected, + 100 * cov$n_units_reporting / pmax(cov$n_units_expected, 1L), + ifelse(cov$is_census_year, "census", "sample") + )) + } + # Explains what the per-row "-- sample year" tag means, regardless of + # which branch above rendered it -- not gated on has_collected, which + # would make this permanently unreachable now that both real callers + # (cog_geographic_rollup(), cog_peer_compare()) always supply it. if (any(!cov$is_census_year)) { cli::cli_text( "Note: the Census of Governments is a complete census only in years ending in 2 or 7; every other year is a sample." diff --git a/R/peers.R b/R/peers.R index 1919e35..851060d 100644 --- a/R/peers.R +++ b/R/peers.R @@ -195,11 +195,13 @@ cog_find_peers <- function(target_govid, #' a balanced panel. #' #' Regardless of mode, `provenance$coverage` always carries per-year -#' `n_units_reporting`, `n_units_expected` and `is_census_year`, and -#' `provenance$coverage_mode` records the mode. `is_census_year` is a -#' statement about the **survey calendar**, never a claim of completeness: -#' FY1967 is a census year in which only 97 of Wisconsin's 608 cities -#' report. `n_units_reporting` is the number that tells the truth. +#' `n_units_expected`, `n_units_collected`, `n_units_reporting` and +#' `is_census_year`, and `provenance$coverage_mode` records the mode. +#' `is_census_year` is a statement about the **survey calendar**, never a +#' claim of completeness: FY1967 is a census year in which only 97 of +#' Wisconsin's 608 cities report. `n_units_reporting` is +#' category-conditional and is not a response rate on its own -- see +#' "Reading `coverage`" below for what each counter answers. #' #' The comparison target is exempt from `"consistent"` balancing -- it is the #' subject of the comparison, not a member of the cohort -- and the @@ -241,14 +243,23 @@ cog_find_peers <- function(target_govid, #' summarise(p50 = quantile(total, 0.5, na.rm = TRUE)) #' ``` #' @section Reading `coverage`: -#' `provenance$coverage` reports `n_units_reporting` against -#' `n_units_expected` per year. **`n_units_reporting` is category-conditional: -#' it counts cohort members with rows for the category you asked for, not -#' cohort members collected that year.** A government that was surveyed and -#' genuinely spends nothing in that category is indistinguishable here from one -#' that was never surveyed. +#' `provenance$coverage` carries three per-year counters: +#' +#' * `n_units_expected` -- how many governments you asked about. +#' * `n_units_collected` -- how many of those appear in the corpus at all +#' that year (in ANY category), separating sampling from real zeros. +#' * `n_units_reporting` -- how many have rows for the SPECIFIC category you +#' requested. This is always <= n_units_collected: a government can be +#' collected but have no rows for "Police" because it contracts policing +#' to the county sheriff, not because it wasn't surveyed. +#' +#' **`n_units_reporting` is category-conditional** and therefore **not a +#' response rate**: `n_units_reporting / n_units_expected` conflates sampling +#' (never collected) with real zeros (collected but spends nothing in your +#' category). Use `n_units_collected / n_units_expected` for the true +#' collection rate, and `n_units_reporting / n_units_collected` for category +#' participation among collected units. #' -#' The ratio is therefore **not a response rate** and must not be used as one. #' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for #' `category = "Police"`; the 174-city gap is overwhelmingly cities that #' contract policing to the county sheriff, not non-response. @@ -325,10 +336,20 @@ cog_peer_compare <- function(target_govid, peers, category, years, # Counted over PEER rows only, against the cohort size: "3 of your 15 peers # reported in FY2019". Including the target would inflate every count by one # and make a cohort that has entirely stopped reporting look non-empty. + # n_units_collected is looked up against the spending long view matching + # whatever basis cog_spending() actually resolved above (prov$basis) -- + # NOT hardcoded to spending_long_harmonized, which does not exist on a + # corpus with schema_version < 5 (R/basis.R resolves basis = "raw" there, + # and only *_long, not *_long_harmonized, is registered; see R/views.R). + # The con comes from .ensure_session() already called inside cog_spending(). + con <- .ensure_session() prov$coverage_mode <- coverage prov$coverage <- .coverage_table( out, years, length(peer_govids), - rows = r[r$role == "peer", , drop = FALSE] + rows = r[r$role == "peer", , drop = FALSE], + con = con, + long_view = .select_long_view("spending_annotated", prov$basis), + expected_ids = peer_govids ) attr(out, "provenance") <- prov out diff --git a/R/rollup.R b/R/rollup.R index 8edac2d..819977e 100644 --- a/R/rollup.R +++ b/R/rollup.R @@ -49,11 +49,13 @@ #' a balanced panel. #' #' Regardless of mode, `provenance$coverage` always carries per-year -#' `n_units_reporting`, `n_units_expected` and `is_census_year`, and -#' `provenance$coverage_mode` records the mode. `is_census_year` is a -#' statement about the **survey calendar**, never a claim of completeness: -#' FY1967 is a census year in which only 97 of Wisconsin's 608 cities -#' report. `n_units_reporting` is the number that tells the truth. +#' `n_units_expected`, `n_units_collected`, `n_units_reporting` and +#' `is_census_year`, and `provenance$coverage_mode` records the mode. +#' `is_census_year` is a statement about the **survey calendar**, never a +#' claim of completeness: FY1967 is a census year in which only 97 of +#' Wisconsin's 608 cities report. `n_units_reporting` is +#' category-conditional and is not a response rate on its own -- see +#' "Reading `coverage`" below for what each counter answers. #' @return Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`, #' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real` / #' `amt_per_capita_nominal` / `amt_per_capita_real`, optional `pop_source`, @@ -61,14 +63,23 @@ #' `provenance` attribute with `verb = "cog_geographic_rollup"`, `layers`, #' and `rollup$included_govids` / `rollup$excluded_govids`. #' @section Reading `coverage`: -#' `provenance$coverage` reports `n_units_reporting` against -#' `n_units_expected` per year. **`n_units_reporting` is category-conditional: -#' it counts governments with rows for the category you asked for, not -#' governments collected that year.** A government that was surveyed and -#' genuinely spends nothing in that category is indistinguishable here from one -#' that was never surveyed. +#' `provenance$coverage` carries three per-year counters: +#' +#' * `n_units_expected` -- how many governments you asked about. +#' * `n_units_collected` -- how many of those appear in the corpus at all +#' that year (in ANY category), separating sampling from real zeros. +#' * `n_units_reporting` -- how many have rows for the SPECIFIC category you +#' requested. This is always <= n_units_collected: a government can be +#' collected but have no rows for "Police" because it contracts policing +#' to the county sheriff, not because it wasn't surveyed. +#' +#' **`n_units_reporting` is category-conditional** and therefore **not a +#' response rate**: `n_units_reporting / n_units_expected` conflates sampling +#' (never collected) with real zeros (collected but spends nothing in your +#' category). Use `n_units_collected / n_units_expected` for the true +#' collection rate, and `n_units_reporting / n_units_collected` for category +#' participation among collected units. #' -#' The ratio is therefore **not a response rate** and must not be used as one. #' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for #' `category = "Police"`; the 174-city gap is overwhelmingly cities that #' contract policing to the county sheriff, not non-response. @@ -137,8 +148,20 @@ cog_geographic_rollup <- function(govids, category, years, # n_units_expected is the universe the CALLER named -- the govids passed in # -- not the national universe. That is what makes the ratio meaningful: # "597 of the 608 Wisconsin cities you asked about reported in FY2012". + # n_units_collected is looked up against the spending long view matching + # whatever basis cog_spending() actually resolved above (prov$basis) -- + # NOT hardcoded to spending_long_harmonized, which does not exist on a + # corpus with schema_version < 5 (R/basis.R resolves basis = "raw" there, + # and only *_long, not *_long_harmonized, is registered; see R/views.R). + # The con comes from .ensure_session() already called inside cog_spending(). + con <- .ensure_session() prov$coverage_mode <- coverage - prov$coverage <- .coverage_table(r, years, length(unique(all_govids))) + prov$coverage <- .coverage_table( + r, years, length(unique(all_govids)), + con = con, + long_view = .select_long_view("spending_annotated", prov$basis), + expected_ids = all_govids + ) attr(r, "provenance") <- prov r diff --git a/README.md b/README.md index c265eb8..9b19cb6 100644 --- a/README.md +++ b/README.md @@ -227,7 +227,8 @@ A statewide total resting on a fifth of the universe looks exactly like one resting on all of it, so every multi-government result now says which it is: ```r -attr(rollup, "provenance")$coverage # per-year n_units_reporting, is_census_year +attr(rollup, "provenance")$coverage +# per-year n_units_expected, n_units_collected, n_units_reporting, is_census_year ``` `cog_geographic_rollup()`, `cog_peer_compare()` and `cog_find_peers()` take a @@ -235,8 +236,14 @@ attr(rollup, "provenance")$coverage # per-year n_units_reporting, is_census_ye `"consistent"` (only units reporting in every requested year, a balanced panel). -`n_units_reporting` is **category-conditional**, and it is not a response rate. A government that was surveyed and genuinely spends -nothing in the requested category is indistinguishable from one never surveyed. +`n_units_reporting` is **category-conditional**: it counts governments with +rows for the *specific* category you asked for, so a government that was +surveyed and genuinely spends nothing in that category is indistinguishable +from one never surveyed — it is not a response rate on its own. +`n_units_collected` is the number that separates them: governments present in +the corpus that year for *any* category. `n_units_collected / n_units_expected` +is the true collection rate; `n_units_reporting / n_units_collected` is +category participation among collected units. ### Absent cells mean two different things diff --git a/man/cog_geographic_rollup.Rd b/man/cog_geographic_rollup.Rd index 15245a6..a24f312 100644 --- a/man/cog_geographic_rollup.Rd +++ b/man/cog_geographic_rollup.Rd @@ -55,11 +55,13 @@ Direct spending); `"primary"` and `"direct"` combine safely.} a balanced panel. Regardless of mode, `provenance$coverage` always carries per-year - `n_units_reporting`, `n_units_expected` and `is_census_year`, and - `provenance$coverage_mode` records the mode. `is_census_year` is a - statement about the **survey calendar**, never a claim of completeness: - FY1967 is a census year in which only 97 of Wisconsin's 608 cities - report. `n_units_reporting` is the number that tells the truth.} + `n_units_expected`, `n_units_collected`, `n_units_reporting` and + `is_census_year`, and `provenance$coverage_mode` records the mode. + `is_census_year` is a statement about the **survey calendar**, never a + claim of completeness: FY1967 is a census year in which only 97 of + Wisconsin's 608 cities report. `n_units_reporting` is + category-conditional and is not a response rate on its own -- see + "Reading `coverage`" below for what each counter answers.} } \value{ Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`, @@ -86,14 +88,23 @@ by design — see `vignette('population-denominators')`. } \section{Reading `coverage`}{ -`provenance$coverage` reports `n_units_reporting` against -`n_units_expected` per year. **`n_units_reporting` is category-conditional: -it counts governments with rows for the category you asked for, not -governments collected that year.** A government that was surveyed and -genuinely spends nothing in that category is indistinguishable here from one -that was never surveyed. +`provenance$coverage` carries three per-year counters: + + * `n_units_expected` -- how many governments you asked about. + * `n_units_collected` -- how many of those appear in the corpus at all + that year (in ANY category), separating sampling from real zeros. + * `n_units_reporting` -- how many have rows for the SPECIFIC category you + requested. This is always <= n_units_collected: a government can be + collected but have no rows for "Police" because it contracts policing + to the county sheriff, not because it wasn't surveyed. + +**`n_units_reporting` is category-conditional** and therefore **not a +response rate**: `n_units_reporting / n_units_expected` conflates sampling +(never collected) with real zeros (collected but spends nothing in your +category). Use `n_units_collected / n_units_expected` for the true +collection rate, and `n_units_reporting / n_units_collected` for category +participation among collected units. -The ratio is therefore **not a response rate** and must not be used as one. In FY2022 — a complete census year — Georgia reports 393 of 567 cities for `category = "Police"`; the 174-city gap is overwhelmingly cities that contract policing to the county sheriff, not non-response. diff --git a/man/cog_peer_compare.Rd b/man/cog_peer_compare.Rd index c918195..b76914f 100644 --- a/man/cog_peer_compare.Rd +++ b/man/cog_peer_compare.Rd @@ -50,11 +50,13 @@ safely.} a balanced panel. Regardless of mode, `provenance$coverage` always carries per-year - `n_units_reporting`, `n_units_expected` and `is_census_year`, and - `provenance$coverage_mode` records the mode. `is_census_year` is a - statement about the **survey calendar**, never a claim of completeness: - FY1967 is a census year in which only 97 of Wisconsin's 608 cities - report. `n_units_reporting` is the number that tells the truth. + `n_units_expected`, `n_units_collected`, `n_units_reporting` and + `is_census_year`, and `provenance$coverage_mode` records the mode. + `is_census_year` is a statement about the **survey calendar**, never a + claim of completeness: FY1967 is a census year in which only 97 of + Wisconsin's 608 cities report. `n_units_reporting` is + category-conditional and is not a response rate on its own -- see + "Reading `coverage`" below for what each counter answers. The comparison target is exempt from `"consistent"` balancing -- it is the subject of the comparison, not a member of the cohort -- and the @@ -109,14 +111,23 @@ them. } \section{Reading `coverage`}{ -`provenance$coverage` reports `n_units_reporting` against -`n_units_expected` per year. **`n_units_reporting` is category-conditional: -it counts cohort members with rows for the category you asked for, not -cohort members collected that year.** A government that was surveyed and -genuinely spends nothing in that category is indistinguishable here from one -that was never surveyed. +`provenance$coverage` carries three per-year counters: + + * `n_units_expected` -- how many governments you asked about. + * `n_units_collected` -- how many of those appear in the corpus at all + that year (in ANY category), separating sampling from real zeros. + * `n_units_reporting` -- how many have rows for the SPECIFIC category you + requested. This is always <= n_units_collected: a government can be + collected but have no rows for "Police" because it contracts policing + to the county sheriff, not because it wasn't surveyed. + +**`n_units_reporting` is category-conditional** and therefore **not a +response rate**: `n_units_reporting / n_units_expected` conflates sampling +(never collected) with real zeros (collected but spends nothing in your +category). Use `n_units_collected / n_units_expected` for the true +collection rate, and `n_units_reporting / n_units_collected` for category +participation among collected units. -The ratio is therefore **not a response rate** and must not be used as one. In FY2022 — a complete census year — Georgia reports 393 of 567 cities for `category = "Police"`; the 174-city gap is overwhelmingly cities that contract policing to the county sheriff, not non-response. diff --git a/tests/testthat/test-coverage-disclosure.R b/tests/testthat/test-coverage-disclosure.R index 979a610..39dff55 100644 --- a/tests/testthat/test-coverage-disclosure.R +++ b/tests/testthat/test-coverage-disclosure.R @@ -48,6 +48,13 @@ test_that("multi-government aggregates disclose reporting coverage on every resu expect_equal(cov$n_units_reporting, c(152L, 597L, 112L, 114L)) expect_equal(cov$is_census_year, c(FALSE, TRUE, FALSE, FALSE)) + # uscogdata#36: with category = NULL (no category scope), "reported at + # all" and "collected" are the same question, so n_units_collected must + # equal n_units_reporting exactly here. This case alone cannot catch a + # regression in HOW n_units_collected is computed, though: see the + # category-scoped test below for that. + expect_equal(cov$n_units_collected, cov$n_units_reporting) + # Cross-check against the raw partitions, scoped to the SAME universe the # rollup was given -- the 608 govids above. Scoping instead on the long # table's own `type`/`fips_state` asks a different question and answers 595: @@ -80,6 +87,8 @@ test_that("multi-government aggregates disclose reporting coverage on every resu expect_equal(cov_peers$n_units_expected, rep(15L, 3L)) expect_equal(cov_peers$n_units_reporting, c(15L, 3L, 3L)) expect_equal(cov_peers$is_census_year, c(TRUE, FALSE, FALSE)) + # uscogdata#36: same identity as the rollup case above, category = NULL. + expect_equal(cov_peers$n_units_collected, cov_peers$n_units_reporting) # -- the three coverage modes -------------------------------------------- expect_equal(attr(cog_peer_compare(target_govid = chilton, peers = peers, @@ -101,3 +110,102 @@ test_that("multi-government aggregates disclose reporting coverage on every resu coverage = "census") expect_equal(sort(unique(census_only$year)), 2012) }) + +test_that("n_units_collected separates sampling from real zeros, category-scoped (uscogdata#36)", { + # The motivating case from the issue: Wisconsin cities, category = "Police". + # FY2012 is a complete census year -- collection is not partial -- yet a + # category-conditional n_units_reporting alone reads like a sampling gap. + # n_units_collected must diverge from n_units_reporting here, unlike the + # category = NULL cases above, because most of the FY2012 gap is cities + # that contract policing to the county sheriff (collected, real zero), not + # cities Census never surveyed. + wi <- cog_gov_search(name = NULL, state = "WI", type = "city") + roll <- suppressMessages(cog_geographic_rollup( + govids = list(city = wi$canonical_govid), category = "Police", + years = c(2011L, 2012L, 2019L, 2020L))) + cov <- wt_coverage(roll) + + expect_equal(cov$n_units_expected, rep(608L, 4L)) + expect_equal(cov$n_units_collected, c(152L, 597L, 112L, 114L)) + expect_equal(cov$n_units_reporting, c(152L, 485L, 109L, 111L)) + + # The pair the issue actually wants: collected/expected is the true + # collection rate (98% in the FY2012 census year, matching the raw + # cross-check above); reporting/collected is category participation among + # collected units (81% -- most of the gap is real, not sampling). + expect_equal(round(cov$n_units_collected[cov$year == 2012] / + cov$n_units_expected[cov$year == 2012], 2), 0.98) + expect_equal(round(cov$n_units_reporting[cov$year == 2012] / + cov$n_units_collected[cov$year == 2012], 2), 0.81) + + # Every year: collected is bounded between reporting and expected. + expect_true(all(cov$n_units_collected >= cov$n_units_reporting)) + expect_true(all(cov$n_units_collected <= cov$n_units_expected)) +}) + +test_that(".coverage_table() candidates a government collected-but-absent from the category result (uscogdata#36)", { + # Direct regression test for the mechanism itself: n_units_collected's + # candidate list must be the caller's full expected cohort (expected_ids), + # never derived from `result`/`rows`. A government with zero rows in the + # requested category across every requested year never appears in + # `result` at all, so deriving candidates from `result` would silently + # drop exactly the "collected but real zero" governments this counter + # exists to count -- collapsing it back to n_units_reporting. + con <- uscogdata:::.ensure_session() + + # A real fixture govid, present in spending_long_harmonized for 2019 (in + # SOME category), but absent from this fake category-specific `result`. + govid <- "011021100004" + fake_result <- data.frame(canonical_govid = character(0), year = integer(0)) + + cov <- uscogdata:::.coverage_table( + fake_result, years = 2019L, n_expected = 1L, + con = con, long_view = "spending_long_harmonized", + expected_ids = govid + ) + expect_equal(cov$n_units_collected, 1L) + expect_equal(cov$n_units_reporting, 0L) + + # Without a connection, long_view, or expected_ids, the lookup is skipped + # rather than silently wrong. + no_con <- uscogdata:::.coverage_table(fake_result, years = 2019L, n_expected = 1L) + expect_true(is.na(no_con$n_units_collected)) + + no_ids <- uscogdata:::.coverage_table( + fake_result, years = 2019L, n_expected = 1L, + con = con, long_view = "spending_long_harmonized" + ) + expect_true(is.na(no_ids$n_units_collected)) +}) + +test_that("n_units_collected uses the resolved basis's long view, not a hardcoded harmonized one (uscogdata#36)", { + # spending_long_harmonized only exists when schema_version >= 5 (R/views.R + # gates the harmonization views on it); on an older corpus cog_spending() + # resolves basis = "raw" and queries spending_long instead. The coverage + # lookup must follow the SAME resolved basis, not a literal + # "spending_long_harmonized", or it hard-errors with a DuckDB catalog + # error on every schema_version < 5 corpus -- a vintage the package + # otherwise explicitly still supports (see test-manifest.R's dual-accept + # tests). + skip_if_no_corpus() + with_doctored_schema_version(4L, { + con <- cog_open() + ids <- DBI::dbGetQuery(con, + "SELECT DISTINCT canonical_govid FROM spending_long WHERE year = 2011 LIMIT 3" + )$canonical_govid + expect_gte(length(ids), 3L) + + roll <- suppressMessages(cog_geographic_rollup( + list(city = ids), category = NULL, years = 2011L)) + expect_equal(attr(roll, "provenance")$basis, "raw") + cov <- attr(roll, "provenance")$coverage + expect_false(is.na(cov$n_units_collected)) + expect_equal(cov$n_units_collected, length(ids)) + + cmp <- suppressMessages(cog_peer_compare( + target_govid = ids[1], peers = ids[-1], category = NULL, years = 2011L)) + expect_equal(attr(cmp, "provenance")$basis, "raw") + cov_peers <- attr(cmp, "provenance")$coverage + expect_false(is.na(cov_peers$n_units_collected)) + }) +}) diff --git a/tests/testthat/test-explain.R b/tests/testthat/test-explain.R index bbcfc41..41566e5 100644 --- a/tests/testthat/test-explain.R +++ b/tests/testthat/test-explain.R @@ -118,3 +118,17 @@ test_that("cog_explain prints denominator + popyear_range + counts", { expect_false(grepl("popyear range: 19-20", out, fixed = TRUE)) }) }) + +test_that("cog_explain reports units collected alongside units reporting (uscogdata#36)", { + skip_if_no_corpus() + wi <- cog_gov_search(name = NULL, state = "WI", type = "city") + roll <- suppressMessages(cog_geographic_rollup( + govids = list(city = wi$canonical_govid), category = "Police", + years = 2012L)) + out <- paste(c( + capture.output(cog_explain(roll)), + capture.output(cog_explain(roll), type = "message") + ), collapse = "\n") + expect_true(grepl("597 of 608 units collected", out, fixed = TRUE)) + expect_true(grepl("485 reporting in this category", out, fixed = TRUE)) +}) -- 2.54.0