diff --git a/NEWS.md b/NEWS.md index 5ef31bb..97c7430 100644 --- a/NEWS.md +++ b/NEWS.md @@ -1,5 +1,39 @@ # uscogdata 0.1.0 (development) +## Multi-government aggregates now disclose their reporting coverage + +* The Census of Governments is a **complete census only in years ending in 2 + and 7**; every other year is a sample, and the sample varies enormously. On + the bundled fixture, Wisconsin's 608-city universe rolls up **597** + governments in FY2012 and **112** in FY2019 — an 18%-to-98% swing the + return value said nothing about, so a statewide total resting on a fifth of + the universe looked exactly like one resting on all of it. +* `cog_geographic_rollup()`, `cog_peer_compare()` and `cog_find_peers()` gain + `coverage`: + + | value | effect | + |---|---| + | `"all"` (default) | every unit that reported that year — unchanged behaviour | + | `"census"` | census years only; aborts if the range holds none rather than returning nothing | + | `"consistent"` | only units reporting in *every* requested year — a balanced panel | + +* **Regardless of mode**, every result now carries `provenance$coverage` with + per-year `n_units_reporting`, `n_units_expected` and `is_census_year`, plus + `provenance$coverage_mode`. `cog_explain()` prints a "Reporting coverage" + section. So the default mode can no longer mislead silently. +* `is_census_year` is a statement about the **survey calendar**, never a claim + of completeness: FY1967 is a census year in which only 97 of Wisconsin's 608 + cities report. `n_units_reporting` is the number that tells the truth. +* On `cog_peer_compare()` the target is exempt from `"consistent"` balancing — + it is the subject of the comparison, not a member of the cohort — and the + `summary_*` quantiles are computed after the filter, so they describe the + cohort actually returned. `n_units_reporting` counts peers only, against the + cohort size. +* On `cog_find_peers()`, `coverage` governs the cohort **vintage** when `year` + is `NULL`: `"census"` snaps to the most recent census year with an observed + population, so a cohort is not built from a sample year in which most of the + candidate universe is absent. + ## `complete = TRUE`: absent cells, labelled with why they are absent * `cog_spending()` and `cog_revenue()` gain `complete`, defaulting to `FALSE` diff --git a/R/coverage.R b/R/coverage.R new file mode 100644 index 0000000..3094f14 --- /dev/null +++ b/R/coverage.R @@ -0,0 +1,107 @@ +# R/coverage.R +# +# Reporting-coverage disclosure for the multi-government verbs (uscogdata#13, +# findings F-020 and F-023). +# +# The Census of Governments is a COMPLETE CENSUS only in years ending in 2 and +# 7. Every other year is a sample, and the sample varies enormously: on the +# bundled fixture, Wisconsin's 608-city universe reports 597 governments in +# FY2012 and 112 in FY2019. Summing "whatever reported" across those years is +# what the verbs have always done -- correctly -- but the return value said +# nothing about it, so a statewide total resting on 18% of the universe looked +# exactly like one resting on 98%. +# +# Owner's settled design: a `coverage` argument selecting WHICH units to +# include, plus always-on metadata saying how many there were either way. The +# principle behind it: using these verbs correctly must not require the caller +# to know the survey calendar. + +# Years ending in 2 or 7 are full censuses of every government; all others are +# samples. +.CENSUS_YEAR_ENDINGS <- c(2L, 7L) + +#' @noRd +.is_census_year <- function(years) { + as.integer(years) %% 10L %in% .CENSUS_YEAR_ENDINGS +} + +#' @noRd +.validate_coverage <- function(coverage) { + tryCatch( + match.arg(coverage, c("all", "census", "consistent")), + error = function(e) { + cli::cli_abort( + "`coverage` must be one of {.val all}, {.val census} or {.val consistent}.", + class = "uscogdata_invalid_coverage", parent = e + ) + } + ) +} + +#' Restrict `years` to census years for `coverage = "census"`. +#' +#' Aborts rather than returning an empty result when the requested range holds +#' no census year: silently handing back zero rows for a query the caller +#' believes they made is the failure mode this whole issue is about. +#' @noRd +.apply_census_years <- function(years, coverage, verb) { + if (!identical(coverage, "census")) return(as.integer(years)) + keep <- as.integer(years)[.is_census_year(years)] + if (length(keep) == 0L) { + cli::cli_abort(c( + "{.code coverage = \"census\"} leaves no years to query.", + x = "None of the requested years end in 2 or 7: {.val {sort(unique(as.integer(years)))}}.", + i = "Census of Governments years ending in 2 or 7 are complete censuses; all others are samples.", + i = "Use {.code coverage = \"all\"} (the default) to keep every requested year, or request a census year." + ), class = "uscogdata_no_census_years") + } + sort(keep) +} + +#' Keep only units that report in EVERY requested year (a balanced panel). +#' +#' `id_col` is the government identifier; `keep_ids` are rows exempt from the +#' filter (the peer-comparison target, which is the subject of the comparison +#' rather than a member of the cohort being balanced). +#' @noRd +.filter_consistent <- function(result, years, id_col = "canonical_govid", + keep_ids = character(0)) { + years <- unique(as.integer(years)) + if (nrow(result) == 0L || length(years) <= 1L) return(result) + ids <- setdiff(unique(result[[id_col]]), c(NA, keep_ids)) + present <- vapply(ids, function(g) { + all(years %in% unique(as.integer(result$year[result[[id_col]] == g]))) + }, logical(1)) + consistent <- c(ids[present], keep_ids) + result[result[[id_col]] %in% consistent | is.na(result[[id_col]]), , + drop = FALSE] +} + +#' Per-year coverage metadata, always attached regardless of mode. +#' +#' Built from the REQUESTED years rather than the years present in the result, +#' so a year in which nothing reported still appears -- with +#' `n_units_reporting = 0`, which is precisely the disclosure a silently +#' missing year fails to make. +#' +#' `n_units_reporting` describes the result the caller actually received, so +#' under `coverage = "consistent"` it reports the balanced count. `is_census_year` +#' is a statement about the SURVEY CALENDAR, never a claim of completeness: +#' FY1967 is a census year in which only 97 of Wisconsin's 608 cities report. +#' `n_units_reporting` is the number that tells the truth. +#' @noRd +.coverage_table <- function(result, years, n_expected, + id_col = "canonical_govid", rows = NULL) { + years <- sort(unique(as.integer(years))) + src <- if (is.null(rows)) result else rows + reporting <- vapply(years, function(y) { + ids <- src[[id_col]][as.integer(src$year) == y] + length(unique(ids[!is.na(ids)])) + }, integer(1)) + tibble::tibble( + year = years, + n_units_reporting = as.integer(reporting), + n_units_expected = rep(as.integer(n_expected), length(years)), + is_census_year = .is_census_year(years) + ) +} diff --git a/R/explain.R b/R/explain.R index 55d2c34..091321d 100644 --- a/R/explain.R +++ b/R/explain.R @@ -118,6 +118,23 @@ cog_explain <- function(result, format = c("print", "list")) { cli::cli_ul(sugg_lines) } + if (!is.null(prov$coverage) && nrow(prov$coverage) > 0L) { + cli::cli_h2("Reporting coverage") + cli::cli_text("Mode: {prov$coverage_mode %||% 'all'}") + cov <- prov$coverage + cli::cli_ul(sprintf( + "%d: %d of %d units reporting (%.0f%%) -- %s year", + cov$year, cov$n_units_reporting, cov$n_units_expected, + 100 * cov$n_units_reporting / pmax(cov$n_units_expected, 1L), + ifelse(cov$is_census_year, "census", "sample") + )) + if (any(!cov$is_census_year)) { + cli::cli_text( + "Note: the Census of Governments is a complete census only in years ending in 2 or 7; every other year is a sample." + ) + } + } + if (isTRUE(prov$completion$applied)) { cli::cli_h2("Completion") cli::cli_text( diff --git a/R/peers.R b/R/peers.R index 59de466..3b71f12 100644 --- a/R/peers.R +++ b/R/peers.R @@ -19,6 +19,13 @@ #' target's population at `year` to produce absolute bounds. If `FALSE`, #' `pop_range` is interpreted as absolute population counts. #' @param max_peers Integer cap on the number of peers returned. +#' @param coverage Survey-cycle handling; see [cog_peer_compare()]. Here it +#' governs the cohort VINTAGE when `year` is `NULL`: `"census"` snaps to the +#' most recent census year with an observed population, so a cohort is not +#' built from a sample year in which most of the candidate universe is +#' absent. `"consistent"` needs a year range, which cohort selection does not +#' have, so it selects like `"all"` and is carried on the result as +#' `attr(x, "coverage")` for [cog_peer_compare()]. #' @return Tibble with columns `canonical_govid`, `gov_name`, `fips_state`, #' `population`, `pop_ratio`, `rank`. The cohort year is attached as #' `attr(x, "cohort_year")`. @@ -29,7 +36,9 @@ cog_find_peers <- function(target_govid, same_state = FALSE, pop_range = c(0.7, 1.3), is_ratio = TRUE, - max_peers = 10L) { + max_peers = 10L, + coverage = c("all", "census", "consistent")) { + coverage <- .validate_coverage(coverage) if (!is.character(target_govid) || length(target_govid) != 1L) { cli::cli_abort("`target_govid` must be a length-1 character string.") } @@ -59,7 +68,7 @@ cog_find_peers <- function(target_govid, )) } - cohort_year <- .resolve_cohort_year(con, target_govid, year) + cohort_year <- .resolve_cohort_year(con, target_govid, year, coverage) pop_sql <- sprintf( "SELECT population FROM gov_population_yearly @@ -107,12 +116,34 @@ cog_find_peers <- function(target_govid, attr(peers, "cohort_year") <- as.integer(cohort_year) attr(peers, "pop_range") <- as.numeric(pop_range) attr(peers, "is_ratio") <- isTRUE(is_ratio) + attr(peers, "coverage") <- coverage + attr(peers, "is_census_year") <- .is_census_year(cohort_year) peers } +# `coverage` picks the cohort vintage when the caller did not name one. +# "census" snaps to the most recent CENSUS year with an observed population, +# so a cohort is not silently built from a sample year in which most of the +# candidate universe is absent. "consistent" is a comparison-time concept -- +# it needs a year RANGE, which cohort selection does not have -- so it selects +# like "all" here and is carried on the result for cog_peer_compare(). #' @noRd -.resolve_cohort_year <- function(con, target_govid, year) { +.resolve_cohort_year <- function(con, target_govid, year, + coverage = "all") { if (!is.null(year)) return(as.integer(year)) + if (identical(coverage, "census")) { + sql <- sprintf( + "SELECT MAX(year) AS y FROM gov_population_yearly + WHERE canonical_govid = %s AND year %% 10 IN (2, 7)", + .sql_lit_chr(target_govid) + ) + y <- DBI::dbGetQuery(con, sql)$y + if (length(y) > 0L && !is.na(y)) return(as.integer(y)) + cli::cli_abort(c( + "{.code coverage = \"census\"} found no census year with an observed population for {target_govid}.", + i = "Pass an explicit {.arg year}, or use {.code coverage = \"all\"}." + ), class = "uscogdata_no_census_years") + } sql <- sprintf( "SELECT MAX(year) AS y FROM gov_population_yearly WHERE canonical_govid = %s", @@ -149,6 +180,31 @@ cog_find_peers <- function(target_govid, #' `"direct"` is accepted; the `"total"` option exists in [cog_spending()] for #' single-government queries but cannot be used here because combining Total #' across peer sets counts intergovernmental transfers twice. +#' @param coverage How to handle the Census of Governments survey cycle, +#' which is a **complete census only in years ending in 2 and 7** -- every +#' other year is a sample, and the sample varies enormously (on the bundled +#' fixture, Wisconsin's 608-city universe reports 597 governments in FY2012 +#' and 112 in FY2019). +#' +#' * `"all"` (default) -- every unit that reported that year. Unchanged +#' behaviour, so existing code keeps working. +#' * `"census"` -- census years only. Aborts if the requested range holds +#' none, rather than silently returning nothing. +#' * `"consistent"` -- only units reporting in *every* requested year, giving +#' a balanced panel. +#' +#' Regardless of mode, `provenance$coverage` always carries per-year +#' `n_units_reporting`, `n_units_expected` and `is_census_year`, and +#' `provenance$coverage_mode` records the mode. `is_census_year` is a +#' statement about the **survey calendar**, never a claim of completeness: +#' FY1967 is a census year in which only 97 of Wisconsin's 608 cities +#' report. `n_units_reporting` is the number that tells the truth. +#' +#' The comparison target is exempt from `"consistent"` balancing -- it is the +#' subject of the comparison, not a member of the cohort -- and the +#' `summary_*` quantiles are computed AFTER the filter, so they describe the +#' cohort actually returned. `n_units_reporting` counts peers only, against +#' the cohort size: "3 of your 15 peers reported in FY2019". #' @return Tibble matching [cog_spending()]'s columns, plus a `role` #' column taking values `"target"`, `"peer"`, `"summary_p25"`, #' `"summary_p50"`, or `"summary_p75"`, `target_rank` (target's rank @@ -186,9 +242,11 @@ cog_find_peers <- function(target_govid, #' @export cog_peer_compare <- function(target_govid, peers, category, years, per_capita = TRUE, adjust_to_year = NULL, - expenditure_concept = c("direct", "total")) { + expenditure_concept = c("direct", "total"), + coverage = c("all", "census", "consistent")) { call <- match.call() expenditure_concept <- match.arg(expenditure_concept) + coverage <- .validate_coverage(coverage) if (identical(expenditure_concept, "total")) { .abort_concept_not_aggregatable("cog_peer_compare") } @@ -211,9 +269,20 @@ cog_peer_compare <- function(target_govid, peers, category, years, peer_govids <- peer_govids[!is.na(peer_govids) & nzchar(peer_govids)] all_govids <- unique(c(target_govid, peer_govids)) + years <- .apply_census_years(years, coverage, "cog_peer_compare") + r <- cog_spending(all_govids, years, category, per_capita, adjust_to_year) r$role <- ifelse(r$canonical_govid == target_govid, "target", "peer") + # The target is exempt from balancing: it is the subject of the comparison, + # not a member of the cohort being balanced, and dropping it would leave a + # peer comparison with nothing to compare. Filtering happens BEFORE the + # quantiles below, so a "consistent" cohort's summary rows describe that + # cohort rather than the unbalanced one. + if (identical(coverage, "consistent")) { + r <- .filter_consistent(r, years, keep_ids = target_govid) + } + value_col <- .peer_value_col(per_capita, adjust_to_year) summary_rows <- .peer_summary_rows(r, value_col) @@ -234,6 +303,14 @@ cog_peer_compare <- function(target_govid, peers, category, years, canonical_govid = target_govid, gov_name = unique(r$gov_name[r$role == "target"]) ) + # Counted over PEER rows only, against the cohort size: "3 of your 15 peers + # reported in FY2019". Including the target would inflate every count by one + # and make a cohort that has entirely stopped reporting look non-empty. + prov$coverage_mode <- coverage + prov$coverage <- .coverage_table( + out, years, length(peer_govids), + rows = r[r$role == "peer", , drop = FALSE] + ) attr(out, "provenance") <- prov out } diff --git a/R/rollup.R b/R/rollup.R index 9c90d53..7534595 100644 --- a/R/rollup.R +++ b/R/rollup.R @@ -31,6 +31,25 @@ #' across multiple layers of government double-counts intergovernmental #' transfers (a state's payment to a school district is the same dollar the #' district reports as its own Direct spending). +#' @param coverage How to handle the Census of Governments survey cycle, +#' which is a **complete census only in years ending in 2 and 7** -- every +#' other year is a sample, and the sample varies enormously (on the bundled +#' fixture, Wisconsin's 608-city universe reports 597 governments in FY2012 +#' and 112 in FY2019). +#' +#' * `"all"` (default) -- every unit that reported that year. Unchanged +#' behaviour, so existing code keeps working. +#' * `"census"` -- census years only. Aborts if the requested range holds +#' none, rather than silently returning nothing. +#' * `"consistent"` -- only units reporting in *every* requested year, giving +#' a balanced panel. +#' +#' Regardless of mode, `provenance$coverage` always carries per-year +#' `n_units_reporting`, `n_units_expected` and `is_census_year`, and +#' `provenance$coverage_mode` records the mode. `is_census_year` is a +#' statement about the **survey calendar**, never a claim of completeness: +#' FY1967 is a census year in which only 97 of Wisconsin's 608 cities +#' report. `n_units_reporting` is the number that tells the truth. #' @return Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`, #' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real` / #' `amt_per_capita_nominal` / `amt_per_capita_real`, optional `pop_source`, @@ -40,9 +59,11 @@ #' @export cog_geographic_rollup <- function(govids, category, years, per_capita = FALSE, adjust_to_year = NULL, - expenditure_concept = c("direct", "total")) { + expenditure_concept = c("direct", "total"), + coverage = c("all", "census", "consistent")) { call <- match.call() expenditure_concept <- match.arg(expenditure_concept) + coverage <- .validate_coverage(coverage) if (identical(expenditure_concept, "total")) { .abort_concept_not_aggregatable("cog_geographic_rollup") } @@ -59,11 +80,20 @@ cog_geographic_rollup <- function(govids, category, years, layer = rep(layer_names, lengths(govids)) ) + # coverage = "census" drops non-census years BEFORE the query rather than + # after: a sample year's rows are not wanted at all, and fetching them only + # to discard them would also let them into the coverage table. + years <- .apply_census_years(years, coverage, "cog_geographic_rollup") + r <- cog_spending(all_govids, years, category, per_capita, adjust_to_year) r <- dplyr::left_join(r, layer_map, by = "canonical_govid", relationship = "many-to-many") r$scope_note <- .rollup_scope_note(r$layer) + if (identical(coverage, "consistent")) { + r <- .filter_consistent(r, years) + } + excluded <- character(0) if (isTRUE(per_capita) && "pop_source" %in% names(r)) { drop <- r$pop_source == "unavailable" @@ -82,6 +112,11 @@ cog_geographic_rollup <- function(govids, category, years, included_govids = included, excluded_govids = excluded ) + # n_units_expected is the universe the CALLER named -- the govids passed in + # -- not the national universe. That is what makes the ratio meaningful: + # "597 of the 608 Wisconsin cities you asked about reported in FY2012". + prov$coverage_mode <- coverage + prov$coverage <- .coverage_table(r, years, length(unique(all_govids))) attr(r, "provenance") <- prov r diff --git a/man/cog_find_peers.Rd b/man/cog_find_peers.Rd index 8a1444c..6de2665 100644 --- a/man/cog_find_peers.Rd +++ b/man/cog_find_peers.Rd @@ -11,7 +11,8 @@ cog_find_peers( same_state = FALSE, pop_range = c(0.7, 1.3), is_ratio = TRUE, - max_peers = 10L + max_peers = 10L, + coverage = c("all", "census", "consistent") ) } \arguments{ @@ -34,6 +35,14 @@ target's population at `year` to produce absolute bounds. If `FALSE`, `pop_range` is interpreted as absolute population counts.} \item{max_peers}{Integer cap on the number of peers returned.} + +\item{coverage}{Survey-cycle handling; see [cog_peer_compare()]. Here it +governs the cohort VINTAGE when `year` is `NULL`: `"census"` snaps to the +most recent census year with an observed population, so a cohort is not +built from a sample year in which most of the candidate universe is +absent. `"consistent"` needs a year range, which cohort selection does not +have, so it selects like `"all"` and is carried on the result as +`attr(x, "coverage")` for [cog_peer_compare()].} } \value{ Tibble with columns `canonical_govid`, `gov_name`, `fips_state`, diff --git a/man/cog_geographic_rollup.Rd b/man/cog_geographic_rollup.Rd index bafdc90..14ffc2f 100644 --- a/man/cog_geographic_rollup.Rd +++ b/man/cog_geographic_rollup.Rd @@ -10,7 +10,8 @@ cog_geographic_rollup( years, per_capita = FALSE, adjust_to_year = NULL, - expenditure_concept = c("direct", "total") + expenditure_concept = c("direct", "total"), + coverage = c("all", "census", "consistent") ) } \arguments{ @@ -35,6 +36,26 @@ single-government queries but cannot be used here because combining Total across multiple layers of government double-counts intergovernmental transfers (a state's payment to a school district is the same dollar the district reports as its own Direct spending).} + +\item{coverage}{How to handle the Census of Governments survey cycle, + which is a **complete census only in years ending in 2 and 7** -- every + other year is a sample, and the sample varies enormously (on the bundled + fixture, Wisconsin's 608-city universe reports 597 governments in FY2012 + and 112 in FY2019). + + * `"all"` (default) -- every unit that reported that year. Unchanged + behaviour, so existing code keeps working. + * `"census"` -- census years only. Aborts if the requested range holds + none, rather than silently returning nothing. + * `"consistent"` -- only units reporting in *every* requested year, giving + a balanced panel. + + Regardless of mode, `provenance$coverage` always carries per-year + `n_units_reporting`, `n_units_expected` and `is_census_year`, and + `provenance$coverage_mode` records the mode. `is_census_year` is a + statement about the **survey calendar**, never a claim of completeness: + FY1967 is a census year in which only 97 of Wisconsin's 608 cities + report. `n_units_reporting` is the number that tells the truth.} } \value{ Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`, diff --git a/man/cog_peer_compare.Rd b/man/cog_peer_compare.Rd index 4725bb7..a7824f6 100644 --- a/man/cog_peer_compare.Rd +++ b/man/cog_peer_compare.Rd @@ -11,7 +11,8 @@ cog_peer_compare( years, per_capita = TRUE, adjust_to_year = NULL, - expenditure_concept = c("direct", "total") + expenditure_concept = c("direct", "total"), + coverage = c("all", "census", "consistent") ) } \arguments{ @@ -33,6 +34,32 @@ population.} `"direct"` is accepted; the `"total"` option exists in [cog_spending()] for single-government queries but cannot be used here because combining Total across peer sets counts intergovernmental transfers twice.} + +\item{coverage}{How to handle the Census of Governments survey cycle, + which is a **complete census only in years ending in 2 and 7** -- every + other year is a sample, and the sample varies enormously (on the bundled + fixture, Wisconsin's 608-city universe reports 597 governments in FY2012 + and 112 in FY2019). + + * `"all"` (default) -- every unit that reported that year. Unchanged + behaviour, so existing code keeps working. + * `"census"` -- census years only. Aborts if the requested range holds + none, rather than silently returning nothing. + * `"consistent"` -- only units reporting in *every* requested year, giving + a balanced panel. + + Regardless of mode, `provenance$coverage` always carries per-year + `n_units_reporting`, `n_units_expected` and `is_census_year`, and + `provenance$coverage_mode` records the mode. `is_census_year` is a + statement about the **survey calendar**, never a claim of completeness: + FY1967 is a census year in which only 97 of Wisconsin's 608 cities + report. `n_units_reporting` is the number that tells the truth. + + The comparison target is exempt from `"consistent"` balancing -- it is the + subject of the comparison, not a member of the cohort -- and the + `summary_*` quantiles are computed AFTER the filter, so they describe the + cohort actually returned. `n_units_reporting` counts peers only, against + the cohort size: "3 of your 15 peers reported in FY2019".} } \value{ Tibble matching [cog_spending()]'s columns, plus a `role` diff --git a/tests/testthat/test-coverage-disclosure.R b/tests/testthat/test-coverage-disclosure.R index 62e97cb..979a610 100644 --- a/tests/testthat/test-coverage-disclosure.R +++ b/tests/testthat/test-coverage-disclosure.R @@ -30,7 +30,6 @@ wt_coverage <- function(x) { } test_that("multi-government aggregates disclose reporting coverage on every result", { - testthat::skip("Blocked on uscogdata#13 (findings F-020, F-023)") # -- F-020: geographic rollups ------------------------------------------- # Wisconsin's city/village universe is 608 governments. On the bundled @@ -49,10 +48,22 @@ test_that("multi-government aggregates disclose reporting coverage on every resu expect_equal(cov$n_units_reporting, c(152L, 597L, 112L, 114L)) expect_equal(cov$is_census_year, c(FALSE, TRUE, FALSE, FALSE)) + # Cross-check against the raw partitions, scoped to the SAME universe the + # rollup was given -- the 608 govids above. Scoping instead on the long + # table's own `type`/`fips_state` asks a different question and answers 595: + # VERNON VILLAGE and WAUKESHA VILLAGE carry type = 3 there (their as-of-year + # identity, when they were townships) while the xwalk lists them as + # govs_type = 2 (their present identity, as villages). Schema v6 made the + # long table's geography present-harmonized and moved as-of-year to the + # *_asof columns, but `type` still reads as-of-year -- see .validate_schema() + # in R/manifest.R. n_units_reporting counts against the requested universe, + # so 597 is the number that answers "how many of the governments I asked + # about reported". raw_2012 <- wt_raw_query(paste0( "SELECT COUNT(DISTINCT canonical_govid) n FROM read_parquet('", wt_corpus_glob(), "') ", - "WHERE type = 2 AND fips_state = 55 AND year = 2012 ", - "AND LEFT(item_code, 1) IN ('E','F','G') AND NOT is_aggregate")) + "WHERE year = 2012 AND LEFT(item_code, 1) IN ('E','F','G') AND NOT is_aggregate ", + "AND canonical_govid IN (", + paste0("'", wi$canonical_govid, "'", collapse = ","), ")")) expect_equal(cov$n_units_reporting[cov$year == 2012], as.integer(raw_2012$n[[1]])) # -- F-023: peer cohorts --------------------------------------------------