Merge pull request 'feat(coverage): n_units_collected separates sampling from real zeros (#36)' (#71) from fix/coverage-collected-36 into main
This commit was merged in pull request #71.
This commit is contained in:
+76
-6
@@ -84,22 +84,92 @@
|
|||||||
#' `n_units_reporting = 0`, which is precisely the disclosure a silently
|
#' `n_units_reporting = 0`, which is precisely the disclosure a silently
|
||||||
#' missing year fails to make.
|
#' missing year fails to make.
|
||||||
#'
|
#'
|
||||||
#' `n_units_reporting` describes the result the caller actually received, so
|
#' Three counters are returned, each answering a different question:
|
||||||
#' under `coverage = "consistent"` it reports the balanced count. `is_census_year`
|
#'
|
||||||
#' is a statement about the SURVEY CALENDAR, never a claim of completeness:
|
#' * `n_units_expected` -- the universe the caller named (govids passed in,
|
||||||
#' FY1967 is a census year in which only 97 of Wisconsin's 608 cities report.
|
#' or peers for cog_peer_compare). "How many governments did you ask
|
||||||
#' `n_units_reporting` is the number that tells the truth.
|
#' about?"
|
||||||
|
#' * `n_units_collected` -- how many of those appear in the corpus at all
|
||||||
|
#' that year, in ANY category. This is a statement about survey collection,
|
||||||
|
#' independent of what was asked for: "of the governments you named, how
|
||||||
|
#' many did Census actually collect data from this year?" It separates
|
||||||
|
#' sampling (not collected) from real zeros (collected but spends nothing
|
||||||
|
#' in your category).
|
||||||
|
#' * `n_units_reporting` -- how many of those appear with rows for the
|
||||||
|
#' SPECIFIC category you requested. This is always <= n_units_collected:
|
||||||
|
#' a government can be collected but have no rows for "Police" because it
|
||||||
|
#' contracts policing to the county sheriff, not because it wasn't
|
||||||
|
#' surveyed.
|
||||||
|
#'
|
||||||
|
#' `n_units_reporting` therefore conflates two very different things: a unit
|
||||||
|
#' that was not collected (sampling) and a unit that was collected but spends
|
||||||
|
#' nothing in that category. The ratio n_units_collected / n_units_expected is
|
||||||
|
#' the true collection rate; n_units_reporting / n_units_collected measures
|
||||||
|
#' category participation among collected units.
|
||||||
|
#'
|
||||||
|
#' `is_census_year` is a statement about the SURVEY CALENDAR, never a claim of
|
||||||
|
#' completeness: FY1967 is a census year in which only 97 of Wisconsin's 608
|
||||||
|
#' cities report. The counters are what tell the truth.
|
||||||
|
#'
|
||||||
|
#' @param con Active DuckDB connection (used to look up n_units_collected).
|
||||||
|
#' @param long_view The verb's own long view, used for the collection query;
|
||||||
|
#' NULL skips the lookup and leaves n_units_collected as NA_integer_.
|
||||||
|
#' @param expected_ids The full EXPECTED cohort (govids the caller named),
|
||||||
|
#' used as the candidate list for the collection query. Required alongside
|
||||||
|
#' `con`/`long_view` for a correct count -- see the note below on why it
|
||||||
|
#' must not be derived from `result`/`rows`. `NULL`, or non-`NULL` but
|
||||||
|
#' empty after dropping `NA`/`""` entries, skips the lookup and leaves
|
||||||
|
#' n_units_collected as NA_integer_.
|
||||||
#' @noRd
|
#' @noRd
|
||||||
.coverage_table <- function(result, years, n_expected,
|
.coverage_table <- function(result, years, n_expected,
|
||||||
id_col = "canonical_govid", rows = NULL) {
|
id_col = "canonical_govid", rows = NULL,
|
||||||
|
con = NULL, long_view = NULL,
|
||||||
|
expected_ids = NULL) {
|
||||||
years <- sort(unique(as.integer(years)))
|
years <- sort(unique(as.integer(years)))
|
||||||
src <- if (is.null(rows)) result else rows
|
src <- if (is.null(rows)) result else rows
|
||||||
reporting <- vapply(years, function(y) {
|
reporting <- vapply(years, function(y) {
|
||||||
ids <- src[[id_col]][as.integer(src$year) == y]
|
ids <- src[[id_col]][as.integer(src$year) == y]
|
||||||
length(unique(ids[!is.na(ids)]))
|
length(unique(ids[!is.na(ids)]))
|
||||||
}, integer(1))
|
}, integer(1))
|
||||||
|
|
||||||
|
# n_units_collected: count EXPECTED cohort members present in the corpus
|
||||||
|
# for ANY category that year, not just the requested one. This separates
|
||||||
|
# sampling (not collected at all) from real zeros (collected but no rows
|
||||||
|
# for this category). Only computed when a connection, long_view, AND
|
||||||
|
# expected_ids are all provided; otherwise NA_integer_.
|
||||||
|
#
|
||||||
|
# The candidate list MUST be expected_ids, not derived from `result`/
|
||||||
|
# `rows`: a government with zero rows in the requested category across
|
||||||
|
# EVERY requested year never appears in `result` at all, so deriving
|
||||||
|
# candidates from it would silently exclude exactly the "collected but
|
||||||
|
# real zero" governments this counter exists to count -- collapsing
|
||||||
|
# n_units_collected back to n_units_reporting for precisely the case #36
|
||||||
|
# was filed over.
|
||||||
|
if (!is.null(con) && !is.null(long_view) && length(expected_ids) > 0L) {
|
||||||
|
cohort_chr <- .sql_lit_chr(unique(expected_ids[!is.na(expected_ids) &
|
||||||
|
nzchar(expected_ids)]))
|
||||||
|
years_lit <- paste(years, collapse = ",")
|
||||||
|
collected_q <- sprintf(
|
||||||
|
"SELECT year, COUNT(DISTINCT canonical_govid) AS n
|
||||||
|
FROM %s
|
||||||
|
WHERE canonical_govid IN (%s)
|
||||||
|
AND year IN (%s)
|
||||||
|
GROUP BY year",
|
||||||
|
long_view, cohort_chr, years_lit
|
||||||
|
)
|
||||||
|
collected_df <- DBI::dbGetQuery(con, collected_q)
|
||||||
|
collected_map <- setNames(collected_df$n, as.integer(collected_df$year))
|
||||||
|
collected <- vapply(years, function(y) {
|
||||||
|
val <- collected_map[as.character(y)]
|
||||||
|
if (is.na(val)) 0L else as.integer(val)
|
||||||
|
}, integer(1))
|
||||||
|
} else {
|
||||||
|
collected <- rep(NA_integer_, length(years))
|
||||||
|
}
|
||||||
|
|
||||||
tibble::tibble(
|
tibble::tibble(
|
||||||
year = years,
|
year = years,
|
||||||
|
n_units_collected = collected,
|
||||||
n_units_reporting = as.integer(reporting),
|
n_units_reporting = as.integer(reporting),
|
||||||
n_units_expected = rep(as.integer(n_expected), length(years)),
|
n_units_expected = rep(as.integer(n_expected), length(years)),
|
||||||
is_census_year = .is_census_year(years)
|
is_census_year = .is_census_year(years)
|
||||||
|
|||||||
+18
@@ -166,12 +166,30 @@ cog_explain <- function(result, format = c("print", "list")) {
|
|||||||
cli::cli_h2("Reporting coverage")
|
cli::cli_h2("Reporting coverage")
|
||||||
cli::cli_text("Mode: {prov$coverage_mode %||% 'all'}")
|
cli::cli_text("Mode: {prov$coverage_mode %||% 'all'}")
|
||||||
cov <- prov$coverage
|
cov <- prov$coverage
|
||||||
|
has_collected <- "n_units_collected" %in% names(cov)
|
||||||
|
if (has_collected) {
|
||||||
|
# Three counters: collected separates sampling from real zeros;
|
||||||
|
# reporting is category-conditional and never a response rate.
|
||||||
|
cli::cli_ul(sprintf(
|
||||||
|
"%d: %d of %d units collected, %d reporting in this category -- %s year",
|
||||||
|
cov$year,
|
||||||
|
cov$n_units_collected,
|
||||||
|
cov$n_units_expected,
|
||||||
|
cov$n_units_reporting,
|
||||||
|
ifelse(cov$is_census_year, "census", "sample")
|
||||||
|
))
|
||||||
|
} else {
|
||||||
cli::cli_ul(sprintf(
|
cli::cli_ul(sprintf(
|
||||||
"%d: %d of %d units reporting (%.0f%%) -- %s year",
|
"%d: %d of %d units reporting (%.0f%%) -- %s year",
|
||||||
cov$year, cov$n_units_reporting, cov$n_units_expected,
|
cov$year, cov$n_units_reporting, cov$n_units_expected,
|
||||||
100 * cov$n_units_reporting / pmax(cov$n_units_expected, 1L),
|
100 * cov$n_units_reporting / pmax(cov$n_units_expected, 1L),
|
||||||
ifelse(cov$is_census_year, "census", "sample")
|
ifelse(cov$is_census_year, "census", "sample")
|
||||||
))
|
))
|
||||||
|
}
|
||||||
|
# Explains what the per-row "-- sample year" tag means, regardless of
|
||||||
|
# which branch above rendered it -- not gated on has_collected, which
|
||||||
|
# would make this permanently unreachable now that both real callers
|
||||||
|
# (cog_geographic_rollup(), cog_peer_compare()) always supply it.
|
||||||
if (any(!cov$is_census_year)) {
|
if (any(!cov$is_census_year)) {
|
||||||
cli::cli_text(
|
cli::cli_text(
|
||||||
"Note: the Census of Governments is a complete census only in years ending in 2 or 7; every other year is a sample."
|
"Note: the Census of Governments is a complete census only in years ending in 2 or 7; every other year is a sample."
|
||||||
|
|||||||
@@ -195,11 +195,13 @@ cog_find_peers <- function(target_govid,
|
|||||||
#' a balanced panel.
|
#' a balanced panel.
|
||||||
#'
|
#'
|
||||||
#' Regardless of mode, `provenance$coverage` always carries per-year
|
#' Regardless of mode, `provenance$coverage` always carries per-year
|
||||||
#' `n_units_reporting`, `n_units_expected` and `is_census_year`, and
|
#' `n_units_expected`, `n_units_collected`, `n_units_reporting` and
|
||||||
#' `provenance$coverage_mode` records the mode. `is_census_year` is a
|
#' `is_census_year`, and `provenance$coverage_mode` records the mode.
|
||||||
#' statement about the **survey calendar**, never a claim of completeness:
|
#' `is_census_year` is a statement about the **survey calendar**, never a
|
||||||
#' FY1967 is a census year in which only 97 of Wisconsin's 608 cities
|
#' claim of completeness: FY1967 is a census year in which only 97 of
|
||||||
#' report. `n_units_reporting` is the number that tells the truth.
|
#' Wisconsin's 608 cities report. `n_units_reporting` is
|
||||||
|
#' category-conditional and is not a response rate on its own -- see
|
||||||
|
#' "Reading `coverage`" below for what each counter answers.
|
||||||
#'
|
#'
|
||||||
#' The comparison target is exempt from `"consistent"` balancing -- it is the
|
#' The comparison target is exempt from `"consistent"` balancing -- it is the
|
||||||
#' subject of the comparison, not a member of the cohort -- and the
|
#' subject of the comparison, not a member of the cohort -- and the
|
||||||
@@ -241,14 +243,23 @@ cog_find_peers <- function(target_govid,
|
|||||||
#' summarise(p50 = quantile(total, 0.5, na.rm = TRUE))
|
#' summarise(p50 = quantile(total, 0.5, na.rm = TRUE))
|
||||||
#' ```
|
#' ```
|
||||||
#' @section Reading `coverage`:
|
#' @section Reading `coverage`:
|
||||||
#' `provenance$coverage` reports `n_units_reporting` against
|
#' `provenance$coverage` carries three per-year counters:
|
||||||
#' `n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
#'
|
||||||
#' it counts cohort members with rows for the category you asked for, not
|
#' * `n_units_expected` -- how many governments you asked about.
|
||||||
#' cohort members collected that year.** A government that was surveyed and
|
#' * `n_units_collected` -- how many of those appear in the corpus at all
|
||||||
#' genuinely spends nothing in that category is indistinguishable here from one
|
#' that year (in ANY category), separating sampling from real zeros.
|
||||||
#' that was never surveyed.
|
#' * `n_units_reporting` -- how many have rows for the SPECIFIC category you
|
||||||
|
#' requested. This is always <= n_units_collected: a government can be
|
||||||
|
#' collected but have no rows for "Police" because it contracts policing
|
||||||
|
#' to the county sheriff, not because it wasn't surveyed.
|
||||||
|
#'
|
||||||
|
#' **`n_units_reporting` is category-conditional** and therefore **not a
|
||||||
|
#' response rate**: `n_units_reporting / n_units_expected` conflates sampling
|
||||||
|
#' (never collected) with real zeros (collected but spends nothing in your
|
||||||
|
#' category). Use `n_units_collected / n_units_expected` for the true
|
||||||
|
#' collection rate, and `n_units_reporting / n_units_collected` for category
|
||||||
|
#' participation among collected units.
|
||||||
#'
|
#'
|
||||||
#' The ratio is therefore **not a response rate** and must not be used as one.
|
|
||||||
#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||||
#' `category = "Police"`; the 174-city gap is overwhelmingly cities that
|
#' `category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||||
#' contract policing to the county sheriff, not non-response.
|
#' contract policing to the county sheriff, not non-response.
|
||||||
@@ -325,10 +336,20 @@ cog_peer_compare <- function(target_govid, peers, category, years,
|
|||||||
# Counted over PEER rows only, against the cohort size: "3 of your 15 peers
|
# Counted over PEER rows only, against the cohort size: "3 of your 15 peers
|
||||||
# reported in FY2019". Including the target would inflate every count by one
|
# reported in FY2019". Including the target would inflate every count by one
|
||||||
# and make a cohort that has entirely stopped reporting look non-empty.
|
# and make a cohort that has entirely stopped reporting look non-empty.
|
||||||
|
# n_units_collected is looked up against the spending long view matching
|
||||||
|
# whatever basis cog_spending() actually resolved above (prov$basis) --
|
||||||
|
# NOT hardcoded to spending_long_harmonized, which does not exist on a
|
||||||
|
# corpus with schema_version < 5 (R/basis.R resolves basis = "raw" there,
|
||||||
|
# and only *_long, not *_long_harmonized, is registered; see R/views.R).
|
||||||
|
# The con comes from .ensure_session() already called inside cog_spending().
|
||||||
|
con <- .ensure_session()
|
||||||
prov$coverage_mode <- coverage
|
prov$coverage_mode <- coverage
|
||||||
prov$coverage <- .coverage_table(
|
prov$coverage <- .coverage_table(
|
||||||
out, years, length(peer_govids),
|
out, years, length(peer_govids),
|
||||||
rows = r[r$role == "peer", , drop = FALSE]
|
rows = r[r$role == "peer", , drop = FALSE],
|
||||||
|
con = con,
|
||||||
|
long_view = .select_long_view("spending_annotated", prov$basis),
|
||||||
|
expected_ids = peer_govids
|
||||||
)
|
)
|
||||||
attr(out, "provenance") <- prov
|
attr(out, "provenance") <- prov
|
||||||
out
|
out
|
||||||
|
|||||||
+36
-13
@@ -49,11 +49,13 @@
|
|||||||
#' a balanced panel.
|
#' a balanced panel.
|
||||||
#'
|
#'
|
||||||
#' Regardless of mode, `provenance$coverage` always carries per-year
|
#' Regardless of mode, `provenance$coverage` always carries per-year
|
||||||
#' `n_units_reporting`, `n_units_expected` and `is_census_year`, and
|
#' `n_units_expected`, `n_units_collected`, `n_units_reporting` and
|
||||||
#' `provenance$coverage_mode` records the mode. `is_census_year` is a
|
#' `is_census_year`, and `provenance$coverage_mode` records the mode.
|
||||||
#' statement about the **survey calendar**, never a claim of completeness:
|
#' `is_census_year` is a statement about the **survey calendar**, never a
|
||||||
#' FY1967 is a census year in which only 97 of Wisconsin's 608 cities
|
#' claim of completeness: FY1967 is a census year in which only 97 of
|
||||||
#' report. `n_units_reporting` is the number that tells the truth.
|
#' Wisconsin's 608 cities report. `n_units_reporting` is
|
||||||
|
#' category-conditional and is not a response rate on its own -- see
|
||||||
|
#' "Reading `coverage`" below for what each counter answers.
|
||||||
#' @return Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`,
|
#' @return Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`,
|
||||||
#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real` /
|
#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real` /
|
||||||
#' `amt_per_capita_nominal` / `amt_per_capita_real`, optional `pop_source`,
|
#' `amt_per_capita_nominal` / `amt_per_capita_real`, optional `pop_source`,
|
||||||
@@ -61,14 +63,23 @@
|
|||||||
#' `provenance` attribute with `verb = "cog_geographic_rollup"`, `layers`,
|
#' `provenance` attribute with `verb = "cog_geographic_rollup"`, `layers`,
|
||||||
#' and `rollup$included_govids` / `rollup$excluded_govids`.
|
#' and `rollup$included_govids` / `rollup$excluded_govids`.
|
||||||
#' @section Reading `coverage`:
|
#' @section Reading `coverage`:
|
||||||
#' `provenance$coverage` reports `n_units_reporting` against
|
#' `provenance$coverage` carries three per-year counters:
|
||||||
#' `n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
#'
|
||||||
#' it counts governments with rows for the category you asked for, not
|
#' * `n_units_expected` -- how many governments you asked about.
|
||||||
#' governments collected that year.** A government that was surveyed and
|
#' * `n_units_collected` -- how many of those appear in the corpus at all
|
||||||
#' genuinely spends nothing in that category is indistinguishable here from one
|
#' that year (in ANY category), separating sampling from real zeros.
|
||||||
#' that was never surveyed.
|
#' * `n_units_reporting` -- how many have rows for the SPECIFIC category you
|
||||||
|
#' requested. This is always <= n_units_collected: a government can be
|
||||||
|
#' collected but have no rows for "Police" because it contracts policing
|
||||||
|
#' to the county sheriff, not because it wasn't surveyed.
|
||||||
|
#'
|
||||||
|
#' **`n_units_reporting` is category-conditional** and therefore **not a
|
||||||
|
#' response rate**: `n_units_reporting / n_units_expected` conflates sampling
|
||||||
|
#' (never collected) with real zeros (collected but spends nothing in your
|
||||||
|
#' category). Use `n_units_collected / n_units_expected` for the true
|
||||||
|
#' collection rate, and `n_units_reporting / n_units_collected` for category
|
||||||
|
#' participation among collected units.
|
||||||
#'
|
#'
|
||||||
#' The ratio is therefore **not a response rate** and must not be used as one.
|
|
||||||
#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||||
#' `category = "Police"`; the 174-city gap is overwhelmingly cities that
|
#' `category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||||
#' contract policing to the county sheriff, not non-response.
|
#' contract policing to the county sheriff, not non-response.
|
||||||
@@ -137,8 +148,20 @@ cog_geographic_rollup <- function(govids, category, years,
|
|||||||
# n_units_expected is the universe the CALLER named -- the govids passed in
|
# n_units_expected is the universe the CALLER named -- the govids passed in
|
||||||
# -- not the national universe. That is what makes the ratio meaningful:
|
# -- not the national universe. That is what makes the ratio meaningful:
|
||||||
# "597 of the 608 Wisconsin cities you asked about reported in FY2012".
|
# "597 of the 608 Wisconsin cities you asked about reported in FY2012".
|
||||||
|
# n_units_collected is looked up against the spending long view matching
|
||||||
|
# whatever basis cog_spending() actually resolved above (prov$basis) --
|
||||||
|
# NOT hardcoded to spending_long_harmonized, which does not exist on a
|
||||||
|
# corpus with schema_version < 5 (R/basis.R resolves basis = "raw" there,
|
||||||
|
# and only *_long, not *_long_harmonized, is registered; see R/views.R).
|
||||||
|
# The con comes from .ensure_session() already called inside cog_spending().
|
||||||
|
con <- .ensure_session()
|
||||||
prov$coverage_mode <- coverage
|
prov$coverage_mode <- coverage
|
||||||
prov$coverage <- .coverage_table(r, years, length(unique(all_govids)))
|
prov$coverage <- .coverage_table(
|
||||||
|
r, years, length(unique(all_govids)),
|
||||||
|
con = con,
|
||||||
|
long_view = .select_long_view("spending_annotated", prov$basis),
|
||||||
|
expected_ids = all_govids
|
||||||
|
)
|
||||||
attr(r, "provenance") <- prov
|
attr(r, "provenance") <- prov
|
||||||
|
|
||||||
r
|
r
|
||||||
|
|||||||
@@ -227,7 +227,8 @@ A statewide total resting on a fifth of the universe looks exactly like one
|
|||||||
resting on all of it, so every multi-government result now says which it is:
|
resting on all of it, so every multi-government result now says which it is:
|
||||||
|
|
||||||
```r
|
```r
|
||||||
attr(rollup, "provenance")$coverage # per-year n_units_reporting, is_census_year
|
attr(rollup, "provenance")$coverage
|
||||||
|
# per-year n_units_expected, n_units_collected, n_units_reporting, is_census_year
|
||||||
```
|
```
|
||||||
|
|
||||||
`cog_geographic_rollup()`, `cog_peer_compare()` and `cog_find_peers()` take a
|
`cog_geographic_rollup()`, `cog_peer_compare()` and `cog_find_peers()` take a
|
||||||
@@ -235,8 +236,14 @@ attr(rollup, "provenance")$coverage # per-year n_units_reporting, is_census_ye
|
|||||||
`"consistent"` (only units reporting in every requested year, a balanced
|
`"consistent"` (only units reporting in every requested year, a balanced
|
||||||
panel).
|
panel).
|
||||||
|
|
||||||
`n_units_reporting` is **category-conditional**, and it is not a response rate. A government that was surveyed and genuinely spends
|
`n_units_reporting` is **category-conditional**: it counts governments with
|
||||||
nothing in the requested category is indistinguishable from one never surveyed.
|
rows for the *specific* category you asked for, so a government that was
|
||||||
|
surveyed and genuinely spends nothing in that category is indistinguishable
|
||||||
|
from one never surveyed — it is not a response rate on its own.
|
||||||
|
`n_units_collected` is the number that separates them: governments present in
|
||||||
|
the corpus that year for *any* category. `n_units_collected / n_units_expected`
|
||||||
|
is the true collection rate; `n_units_reporting / n_units_collected` is
|
||||||
|
category participation among collected units.
|
||||||
|
|
||||||
### Absent cells mean two different things
|
### Absent cells mean two different things
|
||||||
|
|
||||||
|
|||||||
@@ -55,11 +55,13 @@ Direct spending); `"primary"` and `"direct"` combine safely.}
|
|||||||
a balanced panel.
|
a balanced panel.
|
||||||
|
|
||||||
Regardless of mode, `provenance$coverage` always carries per-year
|
Regardless of mode, `provenance$coverage` always carries per-year
|
||||||
`n_units_reporting`, `n_units_expected` and `is_census_year`, and
|
`n_units_expected`, `n_units_collected`, `n_units_reporting` and
|
||||||
`provenance$coverage_mode` records the mode. `is_census_year` is a
|
`is_census_year`, and `provenance$coverage_mode` records the mode.
|
||||||
statement about the **survey calendar**, never a claim of completeness:
|
`is_census_year` is a statement about the **survey calendar**, never a
|
||||||
FY1967 is a census year in which only 97 of Wisconsin's 608 cities
|
claim of completeness: FY1967 is a census year in which only 97 of
|
||||||
report. `n_units_reporting` is the number that tells the truth.}
|
Wisconsin's 608 cities report. `n_units_reporting` is
|
||||||
|
category-conditional and is not a response rate on its own -- see
|
||||||
|
"Reading `coverage`" below for what each counter answers.}
|
||||||
}
|
}
|
||||||
\value{
|
\value{
|
||||||
Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`,
|
Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`,
|
||||||
@@ -86,14 +88,23 @@ by design — see `vignette('population-denominators')`.
|
|||||||
}
|
}
|
||||||
\section{Reading `coverage`}{
|
\section{Reading `coverage`}{
|
||||||
|
|
||||||
`provenance$coverage` reports `n_units_reporting` against
|
`provenance$coverage` carries three per-year counters:
|
||||||
`n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
|
||||||
it counts governments with rows for the category you asked for, not
|
* `n_units_expected` -- how many governments you asked about.
|
||||||
governments collected that year.** A government that was surveyed and
|
* `n_units_collected` -- how many of those appear in the corpus at all
|
||||||
genuinely spends nothing in that category is indistinguishable here from one
|
that year (in ANY category), separating sampling from real zeros.
|
||||||
that was never surveyed.
|
* `n_units_reporting` -- how many have rows for the SPECIFIC category you
|
||||||
|
requested. This is always <= n_units_collected: a government can be
|
||||||
|
collected but have no rows for "Police" because it contracts policing
|
||||||
|
to the county sheriff, not because it wasn't surveyed.
|
||||||
|
|
||||||
|
**`n_units_reporting` is category-conditional** and therefore **not a
|
||||||
|
response rate**: `n_units_reporting / n_units_expected` conflates sampling
|
||||||
|
(never collected) with real zeros (collected but spends nothing in your
|
||||||
|
category). Use `n_units_collected / n_units_expected` for the true
|
||||||
|
collection rate, and `n_units_reporting / n_units_collected` for category
|
||||||
|
participation among collected units.
|
||||||
|
|
||||||
The ratio is therefore **not a response rate** and must not be used as one.
|
|
||||||
In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||||
`category = "Police"`; the 174-city gap is overwhelmingly cities that
|
`category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||||
contract policing to the county sheriff, not non-response.
|
contract policing to the county sheriff, not non-response.
|
||||||
|
|||||||
+23
-12
@@ -50,11 +50,13 @@ safely.}
|
|||||||
a balanced panel.
|
a balanced panel.
|
||||||
|
|
||||||
Regardless of mode, `provenance$coverage` always carries per-year
|
Regardless of mode, `provenance$coverage` always carries per-year
|
||||||
`n_units_reporting`, `n_units_expected` and `is_census_year`, and
|
`n_units_expected`, `n_units_collected`, `n_units_reporting` and
|
||||||
`provenance$coverage_mode` records the mode. `is_census_year` is a
|
`is_census_year`, and `provenance$coverage_mode` records the mode.
|
||||||
statement about the **survey calendar**, never a claim of completeness:
|
`is_census_year` is a statement about the **survey calendar**, never a
|
||||||
FY1967 is a census year in which only 97 of Wisconsin's 608 cities
|
claim of completeness: FY1967 is a census year in which only 97 of
|
||||||
report. `n_units_reporting` is the number that tells the truth.
|
Wisconsin's 608 cities report. `n_units_reporting` is
|
||||||
|
category-conditional and is not a response rate on its own -- see
|
||||||
|
"Reading `coverage`" below for what each counter answers.
|
||||||
|
|
||||||
The comparison target is exempt from `"consistent"` balancing -- it is the
|
The comparison target is exempt from `"consistent"` balancing -- it is the
|
||||||
subject of the comparison, not a member of the cohort -- and the
|
subject of the comparison, not a member of the cohort -- and the
|
||||||
@@ -109,14 +111,23 @@ them.
|
|||||||
}
|
}
|
||||||
\section{Reading `coverage`}{
|
\section{Reading `coverage`}{
|
||||||
|
|
||||||
`provenance$coverage` reports `n_units_reporting` against
|
`provenance$coverage` carries three per-year counters:
|
||||||
`n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
|
||||||
it counts cohort members with rows for the category you asked for, not
|
* `n_units_expected` -- how many governments you asked about.
|
||||||
cohort members collected that year.** A government that was surveyed and
|
* `n_units_collected` -- how many of those appear in the corpus at all
|
||||||
genuinely spends nothing in that category is indistinguishable here from one
|
that year (in ANY category), separating sampling from real zeros.
|
||||||
that was never surveyed.
|
* `n_units_reporting` -- how many have rows for the SPECIFIC category you
|
||||||
|
requested. This is always <= n_units_collected: a government can be
|
||||||
|
collected but have no rows for "Police" because it contracts policing
|
||||||
|
to the county sheriff, not because it wasn't surveyed.
|
||||||
|
|
||||||
|
**`n_units_reporting` is category-conditional** and therefore **not a
|
||||||
|
response rate**: `n_units_reporting / n_units_expected` conflates sampling
|
||||||
|
(never collected) with real zeros (collected but spends nothing in your
|
||||||
|
category). Use `n_units_collected / n_units_expected` for the true
|
||||||
|
collection rate, and `n_units_reporting / n_units_collected` for category
|
||||||
|
participation among collected units.
|
||||||
|
|
||||||
The ratio is therefore **not a response rate** and must not be used as one.
|
|
||||||
In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||||
`category = "Police"`; the 174-city gap is overwhelmingly cities that
|
`category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||||
contract policing to the county sheriff, not non-response.
|
contract policing to the county sheriff, not non-response.
|
||||||
|
|||||||
@@ -48,6 +48,13 @@ test_that("multi-government aggregates disclose reporting coverage on every resu
|
|||||||
expect_equal(cov$n_units_reporting, c(152L, 597L, 112L, 114L))
|
expect_equal(cov$n_units_reporting, c(152L, 597L, 112L, 114L))
|
||||||
expect_equal(cov$is_census_year, c(FALSE, TRUE, FALSE, FALSE))
|
expect_equal(cov$is_census_year, c(FALSE, TRUE, FALSE, FALSE))
|
||||||
|
|
||||||
|
# uscogdata#36: with category = NULL (no category scope), "reported at
|
||||||
|
# all" and "collected" are the same question, so n_units_collected must
|
||||||
|
# equal n_units_reporting exactly here. This case alone cannot catch a
|
||||||
|
# regression in HOW n_units_collected is computed, though: see the
|
||||||
|
# category-scoped test below for that.
|
||||||
|
expect_equal(cov$n_units_collected, cov$n_units_reporting)
|
||||||
|
|
||||||
# Cross-check against the raw partitions, scoped to the SAME universe the
|
# Cross-check against the raw partitions, scoped to the SAME universe the
|
||||||
# rollup was given -- the 608 govids above. Scoping instead on the long
|
# rollup was given -- the 608 govids above. Scoping instead on the long
|
||||||
# table's own `type`/`fips_state` asks a different question and answers 595:
|
# table's own `type`/`fips_state` asks a different question and answers 595:
|
||||||
@@ -80,6 +87,8 @@ test_that("multi-government aggregates disclose reporting coverage on every resu
|
|||||||
expect_equal(cov_peers$n_units_expected, rep(15L, 3L))
|
expect_equal(cov_peers$n_units_expected, rep(15L, 3L))
|
||||||
expect_equal(cov_peers$n_units_reporting, c(15L, 3L, 3L))
|
expect_equal(cov_peers$n_units_reporting, c(15L, 3L, 3L))
|
||||||
expect_equal(cov_peers$is_census_year, c(TRUE, FALSE, FALSE))
|
expect_equal(cov_peers$is_census_year, c(TRUE, FALSE, FALSE))
|
||||||
|
# uscogdata#36: same identity as the rollup case above, category = NULL.
|
||||||
|
expect_equal(cov_peers$n_units_collected, cov_peers$n_units_reporting)
|
||||||
|
|
||||||
# -- the three coverage modes --------------------------------------------
|
# -- the three coverage modes --------------------------------------------
|
||||||
expect_equal(attr(cog_peer_compare(target_govid = chilton, peers = peers,
|
expect_equal(attr(cog_peer_compare(target_govid = chilton, peers = peers,
|
||||||
@@ -101,3 +110,102 @@ test_that("multi-government aggregates disclose reporting coverage on every resu
|
|||||||
coverage = "census")
|
coverage = "census")
|
||||||
expect_equal(sort(unique(census_only$year)), 2012)
|
expect_equal(sort(unique(census_only$year)), 2012)
|
||||||
})
|
})
|
||||||
|
|
||||||
|
test_that("n_units_collected separates sampling from real zeros, category-scoped (uscogdata#36)", {
|
||||||
|
# The motivating case from the issue: Wisconsin cities, category = "Police".
|
||||||
|
# FY2012 is a complete census year -- collection is not partial -- yet a
|
||||||
|
# category-conditional n_units_reporting alone reads like a sampling gap.
|
||||||
|
# n_units_collected must diverge from n_units_reporting here, unlike the
|
||||||
|
# category = NULL cases above, because most of the FY2012 gap is cities
|
||||||
|
# that contract policing to the county sheriff (collected, real zero), not
|
||||||
|
# cities Census never surveyed.
|
||||||
|
wi <- cog_gov_search(name = NULL, state = "WI", type = "city")
|
||||||
|
roll <- suppressMessages(cog_geographic_rollup(
|
||||||
|
govids = list(city = wi$canonical_govid), category = "Police",
|
||||||
|
years = c(2011L, 2012L, 2019L, 2020L)))
|
||||||
|
cov <- wt_coverage(roll)
|
||||||
|
|
||||||
|
expect_equal(cov$n_units_expected, rep(608L, 4L))
|
||||||
|
expect_equal(cov$n_units_collected, c(152L, 597L, 112L, 114L))
|
||||||
|
expect_equal(cov$n_units_reporting, c(152L, 485L, 109L, 111L))
|
||||||
|
|
||||||
|
# The pair the issue actually wants: collected/expected is the true
|
||||||
|
# collection rate (98% in the FY2012 census year, matching the raw
|
||||||
|
# cross-check above); reporting/collected is category participation among
|
||||||
|
# collected units (81% -- most of the gap is real, not sampling).
|
||||||
|
expect_equal(round(cov$n_units_collected[cov$year == 2012] /
|
||||||
|
cov$n_units_expected[cov$year == 2012], 2), 0.98)
|
||||||
|
expect_equal(round(cov$n_units_reporting[cov$year == 2012] /
|
||||||
|
cov$n_units_collected[cov$year == 2012], 2), 0.81)
|
||||||
|
|
||||||
|
# Every year: collected is bounded between reporting and expected.
|
||||||
|
expect_true(all(cov$n_units_collected >= cov$n_units_reporting))
|
||||||
|
expect_true(all(cov$n_units_collected <= cov$n_units_expected))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".coverage_table() candidates a government collected-but-absent from the category result (uscogdata#36)", {
|
||||||
|
# Direct regression test for the mechanism itself: n_units_collected's
|
||||||
|
# candidate list must be the caller's full expected cohort (expected_ids),
|
||||||
|
# never derived from `result`/`rows`. A government with zero rows in the
|
||||||
|
# requested category across every requested year never appears in
|
||||||
|
# `result` at all, so deriving candidates from `result` would silently
|
||||||
|
# drop exactly the "collected but real zero" governments this counter
|
||||||
|
# exists to count -- collapsing it back to n_units_reporting.
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
|
||||||
|
# A real fixture govid, present in spending_long_harmonized for 2019 (in
|
||||||
|
# SOME category), but absent from this fake category-specific `result`.
|
||||||
|
govid <- "011021100004"
|
||||||
|
fake_result <- data.frame(canonical_govid = character(0), year = integer(0))
|
||||||
|
|
||||||
|
cov <- uscogdata:::.coverage_table(
|
||||||
|
fake_result, years = 2019L, n_expected = 1L,
|
||||||
|
con = con, long_view = "spending_long_harmonized",
|
||||||
|
expected_ids = govid
|
||||||
|
)
|
||||||
|
expect_equal(cov$n_units_collected, 1L)
|
||||||
|
expect_equal(cov$n_units_reporting, 0L)
|
||||||
|
|
||||||
|
# Without a connection, long_view, or expected_ids, the lookup is skipped
|
||||||
|
# rather than silently wrong.
|
||||||
|
no_con <- uscogdata:::.coverage_table(fake_result, years = 2019L, n_expected = 1L)
|
||||||
|
expect_true(is.na(no_con$n_units_collected))
|
||||||
|
|
||||||
|
no_ids <- uscogdata:::.coverage_table(
|
||||||
|
fake_result, years = 2019L, n_expected = 1L,
|
||||||
|
con = con, long_view = "spending_long_harmonized"
|
||||||
|
)
|
||||||
|
expect_true(is.na(no_ids$n_units_collected))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("n_units_collected uses the resolved basis's long view, not a hardcoded harmonized one (uscogdata#36)", {
|
||||||
|
# spending_long_harmonized only exists when schema_version >= 5 (R/views.R
|
||||||
|
# gates the harmonization views on it); on an older corpus cog_spending()
|
||||||
|
# resolves basis = "raw" and queries spending_long instead. The coverage
|
||||||
|
# lookup must follow the SAME resolved basis, not a literal
|
||||||
|
# "spending_long_harmonized", or it hard-errors with a DuckDB catalog
|
||||||
|
# error on every schema_version < 5 corpus -- a vintage the package
|
||||||
|
# otherwise explicitly still supports (see test-manifest.R's dual-accept
|
||||||
|
# tests).
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_doctored_schema_version(4L, {
|
||||||
|
con <- cog_open()
|
||||||
|
ids <- DBI::dbGetQuery(con,
|
||||||
|
"SELECT DISTINCT canonical_govid FROM spending_long WHERE year = 2011 LIMIT 3"
|
||||||
|
)$canonical_govid
|
||||||
|
expect_gte(length(ids), 3L)
|
||||||
|
|
||||||
|
roll <- suppressMessages(cog_geographic_rollup(
|
||||||
|
list(city = ids), category = NULL, years = 2011L))
|
||||||
|
expect_equal(attr(roll, "provenance")$basis, "raw")
|
||||||
|
cov <- attr(roll, "provenance")$coverage
|
||||||
|
expect_false(is.na(cov$n_units_collected))
|
||||||
|
expect_equal(cov$n_units_collected, length(ids))
|
||||||
|
|
||||||
|
cmp <- suppressMessages(cog_peer_compare(
|
||||||
|
target_govid = ids[1], peers = ids[-1], category = NULL, years = 2011L))
|
||||||
|
expect_equal(attr(cmp, "provenance")$basis, "raw")
|
||||||
|
cov_peers <- attr(cmp, "provenance")$coverage
|
||||||
|
expect_false(is.na(cov_peers$n_units_collected))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|||||||
@@ -118,3 +118,17 @@ test_that("cog_explain prints denominator + popyear_range + counts", {
|
|||||||
expect_false(grepl("popyear range: 19-20", out, fixed = TRUE))
|
expect_false(grepl("popyear range: 19-20", out, fixed = TRUE))
|
||||||
})
|
})
|
||||||
})
|
})
|
||||||
|
|
||||||
|
test_that("cog_explain reports units collected alongside units reporting (uscogdata#36)", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
wi <- cog_gov_search(name = NULL, state = "WI", type = "city")
|
||||||
|
roll <- suppressMessages(cog_geographic_rollup(
|
||||||
|
govids = list(city = wi$canonical_govid), category = "Police",
|
||||||
|
years = 2012L))
|
||||||
|
out <- paste(c(
|
||||||
|
capture.output(cog_explain(roll)),
|
||||||
|
capture.output(cog_explain(roll), type = "message")
|
||||||
|
), collapse = "\n")
|
||||||
|
expect_true(grepl("597 of 608 units collected", out, fixed = TRUE))
|
||||||
|
expect_true(grepl("485 reporting in this category", out, fixed = TRUE))
|
||||||
|
})
|
||||||
|
|||||||
Reference in New Issue
Block a user