diff --git a/DESCRIPTION b/DESCRIPTION index f91acc5..498ff41 100644 --- a/DESCRIPTION +++ b/DESCRIPTION @@ -31,5 +31,5 @@ Suggests: Config/testthat/edition: 3 VignetteBuilder: knitr RoxygenNote: 7.3.3 -MinCorpusSchema: 3 -MaxCorpusSchema: 3 +MinCorpusSchema: 4 +MaxCorpusSchema: 4 diff --git a/NEWS.md b/NEWS.md index 375f877..2f171b0 100644 --- a/NEWS.md +++ b/NEWS.md @@ -1,5 +1,25 @@ # uscogdata 0.1.0 (development) +## Breaking: corpus schema_version 4 (Phase P canonical ids) + +* The package now requires corpus `schema_version = 4` (`MinCorpusSchema` / + `MaxCorpusSchema` in `DESCRIPTION` are both `4`); older corpora built + against schema 3 are rejected by `cog_open()` with a clear version-mismatch + error. `canonical_govid` is now uniformly 12 characters across every + vintage the corpus covers (previously a mix of 9-char legacy ids and + 12-char FIPS ids depending on source year) — **every hardcoded + `canonical_govid` literal from a pre-Phase-P corpus is now invalid** and + must be re-resolved via `cog_gov_search()` or the new `canonical_alias` + lookup table. `canonical_fips_xwalk` gains four columns + (`legacy_govs_id`, `census_geoid`, `id_source`; `confidence` is renamed to + `pop_confidence`) and a companion `canonical_alias` table ships in the + corpus for mapping legacy/alternate ids onto the current canonical + namespace. The bundled fixture corpus (`inst/extdata/fixture_corpus/`) has + been regenerated against the Phase P publish tree, now ships the full + `canonical_fips_xwalk` and `canonical_alias` master tables alongside the + 2019-2020 long partitions, and is reproducible via + `data-raw/regenerate_fixture_corpus.R`. + ## Clearer errors when `USCOGDATA_URL` is unconfigured or returns non-JSON * `cog_open()` now aborts with the `uscogdata_url_not_configured` error diff --git a/R/search.R b/R/search.R index d2b78d6..4a28eb7 100644 --- a/R/search.R +++ b/R/search.R @@ -126,9 +126,10 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { canonical_govid = character(0), gov_name = character(0), govs_type = integer(0), type_label = character(0), fips_state = character(0), fips_county = character(0), - fips_place = character(0), first_year = integer(0), - last_year = integer(0), population_acs = integer(0), - confidence = character(0) + fips_place = character(0), legacy_govs_id = character(0), + first_year = integer(0), last_year = integer(0), + census_geoid = character(0), population_acs = integer(0), + pop_confidence = character(0), id_source = character(0) ) } diff --git a/R/session.R b/R/session.R index 422469e..1e9f313 100644 --- a/R/session.R +++ b/R/session.R @@ -12,7 +12,7 @@ cog_open <- function(url = .resolve_url(), DBI::dbExecute(con, "INSTALL httpfs; LOAD httpfs;") manifest <- .fetch_or_cache_manifest(url, cache_dir) - .validate_schema(manifest, expected_version = 3L) + .validate_schema(manifest, expected_version = 4L) .validate_scope(manifest) .register_views(con, url, manifest) diff --git a/data-raw/regenerate_fixture_corpus.R b/data-raw/regenerate_fixture_corpus.R new file mode 100644 index 0000000..9e54756 --- /dev/null +++ b/data-raw/regenerate_fixture_corpus.R @@ -0,0 +1,215 @@ +# data-raw/regenerate_fixture_corpus.R +# +# Regenerate inst/extdata/fixture_corpus/ from a cog_pipeline publish tree. +# +# What this does: +# 1. Copies the year=2019 and year=2020 long partitions as-is (byte-for- +# byte) from /data/long/ into the fixture. +# 2. Copies the full canonical_fips_xwalk.parquet, canonical_alias.parquet, +# and summary_categories.parquet metadata tables as-is (these are small +# cross-vintage registries, not partitioned by year, so the fixture +# ships the complete tables rather than a year-scoped subset). +# 3. Resyncs the four reference docs (data_dictionary.md, +# reader-specification.md, README.md, series_breaks.md) from the +# publish tree's docs/. +# 4. Hand-builds manifest.json for just the files the fixture ships, +# following the shape of the previous fixture manifest but with +# schema_version bumped to whatever the source manifest reports, and +# freshly computed sha256 / row_count / size_bytes for every fixture +# file (never copied from the source manifest, since paths and byte +# layout can differ subtly between a full corpus and a fixture). +# +# This is never a manual job: run it whenever cog_pipeline publishes a new +# corpus vintage that the fixture should track. +# +# Usage (from the uscogdata package root): +# Rscript data-raw/regenerate_fixture_corpus.R +# Rscript data-raw/regenerate_fixture_corpus.R /path/to/publish_cache +# +# Or from R: +# source("data-raw/regenerate_fixture_corpus.R") +# regenerate_fixture_corpus(publish_cache_dir = "/path/to/publish_cache") + +regenerate_fixture_corpus <- function( + publish_cache_dir = file.path( + "..", "cog_pipeline", "_targets", "publish_cache" + ), + fixture_dir = file.path("inst", "extdata", "fixture_corpus"), + fixture_years = c(2019L, 2020L)) { + stopifnot( + requireNamespace("digest", quietly = TRUE), + requireNamespace("jsonlite", quietly = TRUE), + requireNamespace("duckdb", quietly = TRUE), + requireNamespace("DBI", quietly = TRUE) + ) + + publish_cache_dir <- normalizePath(publish_cache_dir, mustWork = TRUE) + if (!dir.exists(fixture_dir)) dir.create(fixture_dir, recursive = TRUE) + + source_manifest <- jsonlite::fromJSON( + file.path(publish_cache_dir, "manifest.json"), + simplifyVector = TRUE + ) + + .copy_long_partitions(publish_cache_dir, fixture_dir, fixture_years) + .copy_metadata_parquets(publish_cache_dir, fixture_dir) + .copy_docs(publish_cache_dir, fixture_dir) + + manifest <- .build_fixture_manifest( + fixture_dir, source_manifest, fixture_years + ) + manifest_path <- file.path(fixture_dir, "manifest.json") + writeLines( + jsonlite::toJSON(manifest, auto_unbox = TRUE, pretty = TRUE, null = "null"), + manifest_path + ) + + size_bytes <- sum(file.info( + list.files(fixture_dir, recursive = TRUE, full.names = TRUE) + )$size) + message(sprintf( + "Fixture corpus regenerated at %s (%.2f MB total).", + fixture_dir, size_bytes / 1024^2 + )) + invisible(manifest) +} + +# Copy each requested year's partition directory (just the parquet file +# inside it) from the publish tree into the fixture, as-is. +#' @noRd +.copy_long_partitions <- function(publish_cache_dir, fixture_dir, years) { + for (yr in years) { + part_rel <- file.path("data", "long", sprintf("year=%d", yr), "part-0.parquet") + src <- file.path(publish_cache_dir, part_rel) + dst <- file.path(fixture_dir, part_rel) + if (!file.exists(src)) { + stop(sprintf("Source partition missing: %s", src)) + } + dir.create(dirname(dst), recursive = TRUE, showWarnings = FALSE) + ok <- file.copy(src, dst, overwrite = TRUE) + if (!ok) stop(sprintf("Failed to copy %s -> %s", src, dst)) + } + invisible(NULL) +} + +# Copy the full (not year-scoped) canonical_fips_xwalk, canonical_alias, and +# summary_categories parquet tables. +#' @noRd +.copy_metadata_parquets <- function(publish_cache_dir, fixture_dir) { + files <- c( + "canonical_fips_xwalk.parquet", + "canonical_alias.parquet", + "summary_categories.parquet" + ) + for (f in files) { + src <- file.path(publish_cache_dir, "data", f) + dst <- file.path(fixture_dir, "data", f) + if (!file.exists(src)) { + stop(sprintf("Source metadata file missing: %s", src)) + } + dir.create(dirname(dst), recursive = TRUE, showWarnings = FALSE) + ok <- file.copy(src, dst, overwrite = TRUE) + if (!ok) stop(sprintf("Failed to copy %s -> %s", src, dst)) + } + invisible(NULL) +} + +# Resync the four reference docs shipped alongside the fixture. +#' @noRd +.copy_docs <- function(publish_cache_dir, fixture_dir) { + docs <- c( + "data_dictionary.md", "reader-specification.md", + "README.md", "series_breaks.md" + ) + dst_dir <- file.path(fixture_dir, "docs") + dir.create(dst_dir, recursive = TRUE, showWarnings = FALSE) + for (f in docs) { + src <- file.path(publish_cache_dir, "docs", f) + if (!file.exists(src)) { + stop(sprintf("Source doc missing: %s", src)) + } + ok <- file.copy(src, file.path(dst_dir, f), overwrite = TRUE) + if (!ok) stop(sprintf("Failed to copy doc %s", f)) + } + invisible(NULL) +} + +# Count rows in a parquet file via an ephemeral DuckDB connection. +#' @noRd +.parquet_row_count <- function(path) { + con <- DBI::dbConnect(duckdb::duckdb()) + on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE) + DBI::dbGetQuery(con, sprintf( + "SELECT COUNT(*) AS n FROM read_parquet(%s)", + .sql_quote(path) + ))$n +} + +#' @noRd +.sql_quote <- function(x) paste0("'", gsub("'", "''", x), "'") + +# Hand-build manifest.json following the shape of the previous fixture +# manifest: schema_version / built_at / pipeline_commit / fixture_note / +# data_vintage / scope / schema / files.long_partitions / files.metadata / +# series_breaks_ref / reader_spec_ref. Every sha256 / row_count / size_bytes +# is freshly computed against the files actually written into fixture_dir. +#' @noRd +.build_fixture_manifest <- function(fixture_dir, source_manifest, years) { + long_partitions <- lapply(years, function(yr) { + rel <- file.path("data", "long", sprintf("year=%d", yr), "part-0.parquet") + path <- file.path(fixture_dir, rel) + list( + year = as.integer(yr), + path = gsub("\\\\", "/", rel), + sha256 = digest::digest(path, algo = "sha256", file = TRUE), + row_count = as.integer(.parquet_row_count(path)), + size_bytes = as.integer(file.info(path)$size) + ) + }) + + metadata_files <- c( + "canonical_alias.parquet", + "canonical_fips_xwalk.parquet", + "summary_categories.parquet" + ) + metadata <- lapply(metadata_files, function(f) { + rel <- file.path("data", f) + path <- file.path(fixture_dir, rel) + list( + path = gsub("\\\\", "/", rel), + sha256 = digest::digest(path, algo = "sha256", file = TRUE), + description = f + ) + }) + + list( + schema_version = as.integer(source_manifest$schema_version), + built_at = format(Sys.time(), "%Y-%m-%dT%H:%M:%SZ", tz = "UTC"), + pipeline_commit = source_manifest$pipeline_commit, + fixture_note = paste( + "Two-year (2019-2020) fixture for uscogdata tests. Full corpus", + "available via USCOGDATA_URL. Regenerated for Phase P", + "(schema_version 4, uniformly 12-char canonical_govid) with the full", + "canonical_fips_xwalk master and the new canonical_alias lookup", + "table via data-raw/regenerate_fixture_corpus.R." + ), + data_vintage = source_manifest$data_vintage, + scope = source_manifest$scope, + schema = source_manifest$schema, + files = list( + long_partitions = long_partitions, + metadata = metadata + ), + series_breaks_ref = source_manifest$series_breaks_ref, + reader_spec_ref = source_manifest$reader_spec_ref + ) +} + +if (identical(environment(), globalenv()) && sys.nframe() == 0L) { + args <- commandArgs(trailingOnly = TRUE) + if (length(args) >= 1L) { + regenerate_fixture_corpus(publish_cache_dir = args[[1]]) + } else { + regenerate_fixture_corpus() + } +} diff --git a/inst/extdata/fixture_corpus/data/canonical_alias.parquet b/inst/extdata/fixture_corpus/data/canonical_alias.parquet new file mode 100644 index 0000000..6115b7f Binary files /dev/null and b/inst/extdata/fixture_corpus/data/canonical_alias.parquet differ diff --git a/inst/extdata/fixture_corpus/data/canonical_fips_xwalk.parquet b/inst/extdata/fixture_corpus/data/canonical_fips_xwalk.parquet index f7b8868..fe2aed1 100644 Binary files a/inst/extdata/fixture_corpus/data/canonical_fips_xwalk.parquet and b/inst/extdata/fixture_corpus/data/canonical_fips_xwalk.parquet differ diff --git a/inst/extdata/fixture_corpus/data/long/year=2019/part-0.parquet b/inst/extdata/fixture_corpus/data/long/year=2019/part-0.parquet index 5e10a35..127794c 100644 Binary files a/inst/extdata/fixture_corpus/data/long/year=2019/part-0.parquet and b/inst/extdata/fixture_corpus/data/long/year=2019/part-0.parquet differ diff --git a/inst/extdata/fixture_corpus/data/long/year=2020/part-0.parquet b/inst/extdata/fixture_corpus/data/long/year=2020/part-0.parquet index f0f8075..4129531 100644 Binary files a/inst/extdata/fixture_corpus/data/long/year=2020/part-0.parquet and b/inst/extdata/fixture_corpus/data/long/year=2020/part-0.parquet differ diff --git a/inst/extdata/fixture_corpus/data/summary_categories.parquet b/inst/extdata/fixture_corpus/data/summary_categories.parquet index 8003ca4..b58d39d 100644 Binary files a/inst/extdata/fixture_corpus/data/summary_categories.parquet and b/inst/extdata/fixture_corpus/data/summary_categories.parquet differ diff --git a/inst/extdata/fixture_corpus/manifest.json b/inst/extdata/fixture_corpus/manifest.json index 600d1b0..9b87d9b 100644 --- a/inst/extdata/fixture_corpus/manifest.json +++ b/inst/extdata/fixture_corpus/manifest.json @@ -1,54 +1,21 @@ { - "schema_version": 3, - "built_at": "2026-04-29T18:57:18Z", - "pipeline_commit": "bd3e744", - "fixture_note": "Two-year (2019-2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated with Layer 1 (canonical_govid resolver gate) + Layer 2 (PID-era extended xwalk).", + "schema_version": 4, + "built_at": "2026-07-11T13:24:01Z", + "pipeline_commit": "1a00925", + "fixture_note": "Two-year (2019-2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated for Phase P (schema_version 4, uniformly 12-char canonical_govid) with the full canonical_fips_xwalk master and the new canonical_alias lookup table via data-raw/regenerate_fixture_corpus.R.", "data_vintage": { "census_source_downloaded": "unknown", "cpi_vintage": "FRED CPIAUCSL", "acs_vintage": "ACS 2018-2022 5-year" }, "scope": { - "gov_types_included": [ - 0, - 1, - 2, - 3 - ], - "gov_types_excluded": [ - 4, - 5 - ], + "gov_types_included": [0, 1, 2, 3], + "gov_types_excluded": [4, 5], "scope_note": "v0.1 covers state, county, city/municipality, and township governments. Special districts (type 4) and school districts (type 5) are excluded pending validation in a future cycle." }, "schema": { "long_column_count": 24, - "long_columns": [ - "fips_state", - "type", - "fips_county", - "govid", - "gov_blank", - "gov_name", - "county_name", - "fips_state_code", - "fips_county_code", - "fips_place_code", - "population", - "popyear", - "enrollment", - "enrollyear", - "function_code", - "sch_level_code", - "fiscal_year_end", - "srvy_year", - "item_code", - "amt", - "srv_data", - "impute_flag", - "is_aggregate", - "canonical_govid" - ], + "long_columns": ["fips_state", "type", "fips_county", "govid", "gov_blank", "gov_name", "county_name", "fips_state_code", "fips_county_code", "fips_place_code", "population", "popyear", "enrollment", "enrollyear", "function_code", "sch_level_code", "fiscal_year_end", "srvy_year", "item_code", "amt", "srv_data", "impute_flag", "is_aggregate", "canonical_govid"], "data_dictionary": "docs/data_dictionary.md" }, "files": { @@ -56,27 +23,32 @@ { "year": 2019, "path": "data/long/year=2019/part-0.parquet", - "sha256": "d93affbbf9c46bd193fa5e07d89dfa9b08ad4f6fc2442fe7a27e6e4ef28f8bd9", + "sha256": "c0a2bf0758af129d5dfddb6ff6665cc435ddee87fd6879e788fb56ed53ab22b8", "row_count": 318139, - "size_bytes": 1425125 + "size_bytes": 1441404 }, { "year": 2020, "path": "data/long/year=2020/part-0.parquet", - "sha256": "c9224833bd914cc85efb2ec62f83f75c34dfa469d4398f5cb1789fd028c03389", + "sha256": "92570b9d55ec3425d034db37838f91c3b8359d0454d3d98730a6016b62e4bb48", "row_count": 317500, - "size_bytes": 1428049 + "size_bytes": 1444011 } ], "metadata": [ + { + "path": "data/canonical_alias.parquet", + "sha256": "db784d4ec9e5abb4b033405627d8bdd6f8d8ff96b3033a5b2587b4b59113ec9c", + "description": "canonical_alias.parquet" + }, { "path": "data/canonical_fips_xwalk.parquet", - "sha256": "4bbdf0415b0ae1a24927869bb9e8e75eca5d7bcceebdf8c26346c7cbf8472d34", + "sha256": "c0b1295e779b601f2782de40ab143556324e74d30d214fc10997c2244c109389", "description": "canonical_fips_xwalk.parquet" }, { "path": "data/summary_categories.parquet", - "sha256": "60045e22bc2723318fa2cb73f8e5038250dc54d24b3447c6750dfe29035335b8", + "sha256": "dd59e7f58a022679ad43511c8c8e938b8dd4be81196bbeeee21e67bdcca2295b", "description": "summary_categories.parquet" } ] diff --git a/tests/testthat/test-explain.R b/tests/testthat/test-explain.R index 61b85b8..83f18ab 100644 --- a/tests/testthat/test-explain.R +++ b/tests/testthat/test-explain.R @@ -1,6 +1,6 @@ test_that("cog_explain prints verb header and target", { skip_if_no_corpus() - r <- cog_spending("101006006", 2020L, "Corrections") + r <- cog_spending("121011212191", 2020L, "Corrections") # cli writes to stderr; capture both stdout and message streams. txt <- paste(c( capture.output(cog_explain(r)), @@ -8,19 +8,19 @@ test_that("cog_explain prints verb header and target", { ), collapse = "\n") expect_true(grepl("cog_spending", txt)) expect_true(grepl("Corrections", txt)) - expect_true(grepl("101006006", txt)) + expect_true(grepl("121011212191", txt)) }) test_that("cog_explain format='list' returns structured provenance", { skip_if_no_corpus() - r <- cog_spending("101006006", 2020L, "Corrections") + r <- cog_spending("121011212191", 2020L, "Corrections") prov <- cog_explain(r, format = "list") expect_identical(prov, attr(r, "provenance")) }) test_that("cog_explain returns result invisibly for chaining", { skip_if_no_corpus() - r <- cog_spending("101006006", 2020L, "Corrections") + r <- cog_spending("121011212191", 2020L, "Corrections") res <- withVisible(cog_explain(r)) expect_false(res$visible) expect_identical(res$value, r) @@ -34,7 +34,7 @@ test_that("cog_explain errors on non-verb input", { test_that("cog_explain prints denominator + popyear_range + counts", { skip_if_no_corpus() with_fixture_corpus({ - r <- cog_spending("101006006", years = 2019:2020, + r <- cog_spending("121011212191", years = 2019:2020, category = "Police", per_capita = TRUE) out <- paste(c( capture.output(cog_explain(r)), diff --git a/tests/testthat/test-mirror.R b/tests/testthat/test-mirror.R index 6e812cb..2e6bac4 100644 --- a/tests/testthat/test-mirror.R +++ b/tests/testthat/test-mirror.R @@ -61,7 +61,7 @@ test_that("cog_mirror reads back via a fresh session against the mirror", { cog_close() options(uscogdata.url = paste0(normalizePath(tmp), "/")) - r <- cog_spending("101006006", 2020L, "Corrections") + r <- cog_spending("121011212191", 2020L, "Corrections") expect_gt(nrow(r), 0L) - expect_equal(unique(r$canonical_govid), "101006006") + expect_equal(unique(r$canonical_govid), "121011212191") }) diff --git a/tests/testthat/test-peers.R b/tests/testthat/test-peers.R index 7dc05e6..2044dc6 100644 --- a/tests/testthat/test-peers.R +++ b/tests/testthat/test-peers.R @@ -1,25 +1,25 @@ test_that("cog_find_peers returns same-type peers in the default pop band", { skip_if_no_corpus() - peers <- cog_find_peers("101006006") # Broward County + peers <- cog_find_peers("121011212191") # Broward County expect_s3_class(peers, "tbl_df") expected_cols <- c("canonical_govid", "gov_name", "fips_state", "population", "pop_ratio", "rank") expect_true(all(expected_cols %in% names(peers))) expect_true(all(peers$pop_ratio >= 0.7 & peers$pop_ratio <= 1.3)) - expect_false("101006006" %in% peers$canonical_govid) + expect_false("121011212191" %in% peers$canonical_govid) expect_equal(peers$rank, seq_len(nrow(peers))) }) test_that("cog_find_peers respects same_state restriction", { skip_if_no_corpus() - peers <- cog_find_peers("101006006", same_state = TRUE, + peers <- cog_find_peers("121011212191", same_state = TRUE, pop_range = c(0.1, 10)) expect_true(all(peers$fips_state == "12")) }) test_that("cog_find_peers absolute pop range works", { skip_if_no_corpus() - peers <- cog_find_peers("101006006", + peers <- cog_find_peers("121011212191", pop_range = c(1.5e6, 2.5e6), is_ratio = FALSE, max_peers = 20L) expect_true(all(peers$population >= 1.5e6 & @@ -33,8 +33,8 @@ test_that("cog_find_peers errors cleanly on unknown govid", { test_that("cog_peer_compare accepts a cog_find_peers result directly", { skip_if_no_corpus() - peers <- cog_find_peers("101006006", max_peers = 4L) - r <- cog_peer_compare("101006006", peers, "Police", years = 2020L) + peers <- cog_find_peers("121011212191", max_peers = 4L) + r <- cog_peer_compare("121011212191", peers, "Police", years = 2020L) expect_s3_class(r, "tbl_df") expect_true("role" %in% names(r)) expect_setequal( @@ -47,8 +47,8 @@ test_that("cog_peer_compare accepts a cog_find_peers result directly", { test_that("cog_peer_compare accepts a character vector of govids", { skip_if_no_corpus() r <- cog_peer_compare( - "101006006", - peers = c("441015015", "441220220"), # Bexar, Tarrant + "121011212191", + peers = c("481029175853", "481439135072"), # Bexar, Tarrant category = "Police", years = 2020L ) expect_true("peer" %in% r$role) @@ -58,8 +58,8 @@ test_that("cog_peer_compare accepts a character vector of govids", { test_that("cog_peer_compare summary rows use real per-capita when requested", { skip_if_no_corpus() r <- cog_peer_compare( - "101006006", - peers = c("441015015", "441220220", "231082082"), + "121011212191", + peers = c("481029175853", "481439135072", "261163166615"), category = "Police", years = 2019:2020, per_capita = TRUE, adjust_to_year = 2022L ) @@ -72,8 +72,8 @@ test_that("cog_peer_compare summary rows use real per-capita when requested", { test_that("cog_peer_compare provenance reports the outer verb + peer count", { skip_if_no_corpus() - r <- cog_peer_compare("101006006", - peers = c("441015015", "441220220"), + r <- cog_peer_compare("121011212191", + peers = c("481029175853", "481439135072"), category = "Police", years = 2020L) prov <- attr(r, "provenance") expect_equal(prov$verb, "cog_peer_compare") @@ -82,7 +82,7 @@ test_that("cog_peer_compare provenance reports the outer verb + peer count", { test_that("cog_peer_compare handles zero peers gracefully", { skip_if_no_corpus() - r <- cog_peer_compare("101006006", + r <- cog_peer_compare("121011212191", peers = character(0), category = "Police", years = 2020L) expect_true(all(r$role == "target")) @@ -91,7 +91,7 @@ test_that("cog_peer_compare handles zero peers gracefully", { test_that("cog_find_peers defaults `year` to most recent observed year for target", { skip_if_no_corpus() - peers <- cog_find_peers("101006006") + peers <- cog_find_peers("121011212191") expect_equal(attr(peers, "cohort_year"), 2020L) # Returned column is now `population`, not `population_acs` expect_true("population" %in% names(peers)) @@ -100,23 +100,23 @@ test_that("cog_find_peers defaults `year` to most recent observed year for targe test_that("cog_find_peers honors an explicit `year`", { skip_if_no_corpus() - peers <- cog_find_peers("101006006", year = 2019L) + peers <- cog_find_peers("121011212191", year = 2019L) expect_equal(attr(peers, "cohort_year"), 2019L) }) test_that("cog_find_peers errors when target has no observed pop in `year`", { skip_if_no_corpus() expect_error( - cog_find_peers("101006006", year = 1999L), + cog_find_peers("121011212191", year = 1999L), "no observed population" ) }) test_that("cog_peer_compare stamps cohort_year from peers attribute", { skip_if_no_corpus() - peers <- cog_find_peers("101006006", year = 2019L, max_peers = 4L, + peers <- cog_find_peers("121011212191", year = 2019L, max_peers = 4L, pop_range = c(0.5, 1.5)) - r <- cog_peer_compare("101006006", peers, "Police", years = 2020L) + r <- cog_peer_compare("121011212191", peers, "Police", years = 2020L) expect_true("cohort_year" %in% names(r)) expect_true(all(r$cohort_year == 2019L)) prov <- attr(r, "provenance") @@ -130,8 +130,8 @@ test_that("cog_peer_compare stamps cohort_year from peers attribute", { test_that("cog_peer_compare cohort_year is NA for bare character peers", { skip_if_no_corpus() r <- cog_peer_compare( - "101006006", - peers = c("441015015", "441220220"), + "121011212191", + peers = c("481029175853", "481439135072"), category = "Police", years = 2020L ) expect_true(all(is.na(r$cohort_year))) diff --git a/tests/testthat/test-revenue.R b/tests/testthat/test-revenue.R index 7c28290..a979996 100644 --- a/tests/testthat/test-revenue.R +++ b/tests/testthat/test-revenue.R @@ -1,24 +1,24 @@ test_that("cog_revenue returns expected shape for Broward Property Tax 2020", { skip_if_no_corpus() - r <- cog_revenue("101006006", years = 2020L, category = "Property Tax") + r <- cog_revenue("121011212191", years = 2020L, category = "Property Tax") expect_s3_class(r, "tbl_df") expected_cols <- c("year", "canonical_govid", "gov_name", "revenue_subtype", "category", "amt_nominal", "codes_included", "aggregate_fallback", "notes") expect_true(all(expected_cols %in% names(r))) - expect_equal(unique(r$canonical_govid), "101006006") + expect_equal(unique(r$canonical_govid), "121011212191") expect_equal(unique(r$year), 2020L) }) test_that("cog_revenue with no category filter returns multiple categories", { skip_if_no_corpus() - r <- cog_revenue("101006006", years = 2020L) + r <- cog_revenue("121011212191", years = 2020L) expect_gt(length(unique(r$category)), 1L) }) test_that("cog_revenue with per_capita + adjust_to_year adds all columns", { skip_if_no_corpus() - r <- cog_revenue("101006006", 2020L, + r <- cog_revenue("121011212191", 2020L, per_capita = TRUE, adjust_to_year = 2022L) expect_true(all(c("amt_nominal", "amt_real", "amt_per_capita_nominal", "amt_per_capita_real") %in% @@ -27,7 +27,7 @@ test_that("cog_revenue with per_capita + adjust_to_year adds all columns", { test_that("cog_revenue result has provenance attribute", { skip_if_no_corpus() - r <- cog_revenue("101006006", 2020L) + r <- cog_revenue("121011212191", 2020L) prov <- attr(r, "provenance") expect_equal(prov$verb, "cog_revenue") expect_true(grepl("revenue_annotated", prov$sql_query)) diff --git a/tests/testthat/test-rollup.R b/tests/testthat/test-rollup.R index 424ae72..0caaec9 100644 --- a/tests/testthat/test-rollup.R +++ b/tests/testthat/test-rollup.R @@ -2,9 +2,9 @@ test_that("cog_geographic_rollup aggregates state + county + city layers", { skip_if_no_corpus() r <- cog_geographic_rollup( govids = list( - state = "100000000", # Florida state govt - county = "101006006", # Broward County - city = "102006004" # Fort Lauderdale City + state = "120000226351", # Florida state govt + county = "121011212191", # Broward County + city = "122011161585" # Fort Lauderdale City ), category = "Police", years = 2019:2020 @@ -23,7 +23,7 @@ test_that("cog_geographic_rollup aggregates state + county + city layers", { test_that("cog_geographic_rollup respects per_capita + adjust_to_year", { skip_if_no_corpus() r <- cog_geographic_rollup( - govids = list(county = "101006006", city = "102006004"), + govids = list(county = "121011212191", city = "122011161585"), category = "Police", years = 2020L, per_capita = TRUE, @@ -43,8 +43,8 @@ test_that("cog_geographic_rollup respects per_capita + adjust_to_year", { test_that("cog_geographic_rollup scope_notes describe each layer", { skip_if_no_corpus() r <- cog_geographic_rollup( - govids = list(state = "100000000", county = "101006006", - city = "102006004"), + govids = list(state = "120000226351", county = "121011212191", + city = "122011161585"), category = "Police", years = 2020L ) state_notes <- unique(r$scope_note[r$layer == "state"]) @@ -58,7 +58,7 @@ test_that("cog_geographic_rollup scope_notes describe each layer", { test_that("cog_geographic_rollup single-layer call works", { skip_if_no_corpus() r <- cog_geographic_rollup( - govids = list(county = c("101006006")), + govids = list(county = c("121011212191")), category = "Corrections", years = 2020L ) @@ -69,7 +69,7 @@ test_that("cog_geographic_rollup single-layer call works", { test_that("cog_geographic_rollup provenance reports the outer verb", { skip_if_no_corpus() r <- cog_geographic_rollup( - govids = list(state = "100000000", county = "101006006"), + govids = list(state = "120000226351", county = "121011212191"), category = "Police", years = 2020L ) prov <- attr(r, "provenance") @@ -80,7 +80,7 @@ test_that("cog_geographic_rollup provenance reports the outer verb", { test_that("cog_geographic_rollup accepts data.frames per layer", { skip_if_no_corpus() - fl_state <- cog_gov_search("^FLORIDA STATE GOVT$", type = "state") + fl_state <- cog_gov_search("^FLORIDA$", type = "state") broward <- cog_gov_search("^BROWARD COUNTY$", state = "FL", type = "county") r <- cog_geographic_rollup( govids = list(state = fl_state, county = broward), @@ -92,9 +92,9 @@ test_that("cog_geographic_rollup accepts data.frames per layer", { test_that("cog_geographic_rollup rejects invalid inputs", { expect_error(cog_geographic_rollup(list(), "Police", 2020L), "length") - expect_error(cog_geographic_rollup(c("101006006"), "Police", 2020L), "list") + expect_error(cog_geographic_rollup(c("121011212191"), "Police", 2020L), "list") expect_error( - cog_geographic_rollup(list(planet = "100000000"), "Police", 2020L), + cog_geographic_rollup(list(planet = "120000226351"), "Police", 2020L), "state|county|city" ) }) @@ -103,8 +103,8 @@ test_that("cog_geographic_rollup per-capita uses summed per-year populations", { skip_if_no_corpus() with_fixture_corpus({ r <- cog_geographic_rollup( - govids = list(state = "010000000", - county = "101006006"), + govids = list(state = "010000226085", + county = "121011212191"), category = "Police", years = 2019:2020, per_capita = TRUE @@ -125,14 +125,14 @@ test_that("cog_geographic_rollup records included/excluded govids in provenance" skip_if_no_corpus() with_fixture_corpus({ r <- cog_geographic_rollup( - govids = list(county = "101006006"), + govids = list(county = "121011212191"), category = "Police", years = 2019:2020, per_capita = TRUE ) prov <- attr(r, "provenance") expect_true("rollup" %in% names(prov)) - expect_true("101006006" %in% prov$rollup$included_govids) + expect_true("121011212191" %in% prov$rollup$included_govids) expect_true(is.character(prov$rollup$excluded_govids)) }) }) diff --git a/tests/testthat/test-search.R b/tests/testthat/test-search.R index 07bdb7e..adb4959 100644 --- a/tests/testthat/test-search.R +++ b/tests/testthat/test-search.R @@ -146,7 +146,7 @@ test_that(".resolve_basket_row exact match returns one row", { expect_equal(out$match_method, "exact") expect_equal(out$n_candidates, 1L) expect_equal(nrow(out$row), 1L) - expect_equal(out$row$canonical_govid, "101006006") + expect_equal(out$row$canonical_govid, "121011212191") expect_equal(out$row$gov_name, "BROWARD COUNTY") }) @@ -157,7 +157,7 @@ test_that(".resolve_basket_row exact match is case-insensitive", { ) expect_equal(out$status, "resolved") expect_equal(out$match_method, "exact") - expect_equal(out$row$canonical_govid, "101006006") + expect_equal(out$row$canonical_govid, "121011212191") }) test_that(".resolve_basket_row exact match honors per-row type", { @@ -166,7 +166,7 @@ test_that(".resolve_basket_row exact match honors per-row type", { name = "SAN DIEGO CITY", state = "CA", type = "city", con = con ) expect_equal(out$status, "resolved") - expect_equal(out$row$canonical_govid, "052037010") + expect_equal(out$row$canonical_govid, "062073207598") }) test_that(".resolve_basket_row substring fallback resolves single match", { @@ -177,7 +177,7 @@ test_that(".resolve_basket_row substring fallback resolves single match", { expect_equal(out$status, "resolved") expect_equal(out$match_method, "substring") expect_equal(out$n_candidates, 1L) - expect_equal(out$row$canonical_govid, "101006006") + expect_equal(out$row$canonical_govid, "121011212191") }) test_that(".resolve_basket_row no_match returns 0-row tibble", { @@ -207,15 +207,19 @@ test_that(".resolve_basket_row treats empty/whitespace name as no_match", { test_that(".resolve_basket_row largest_pop within single type", { # FL Miami substring matches 10 cities (all govs_type = 2), largest pop - # is MIAMI CITY at 443665. + # is MIAMI CITY at 443665. Under Phase P canonical naming, MIAMI-DADE + # COUNTY (govs_type = 1) also contains "Miami", so `type = "city"` pins + # the match set to a single type (as the query docs promise it will for + # per-row `type`), keeping this test's original intent: multiple + # same-type name matches resolve to the largest-population row. con <- uscogdata:::.ensure_session() out <- uscogdata:::.resolve_basket_row( - name = "Miami", state = "FL", type = NA_character_, con = con + name = "Miami", state = "FL", type = "city", con = con ) expect_equal(out$status, "largest_pop") expect_equal(out$match_method, "substring") expect_gte(out$n_candidates, 2L) - expect_equal(out$row$canonical_govid, "102013013") + expect_equal(out$row$canonical_govid, "122086194757") expect_equal(out$row$gov_name, "MIAMI CITY") }) @@ -241,7 +245,7 @@ test_that(".resolve_basket_row resolves with type override on ambiguous case", { ) expect_equal(out$status, "resolved") expect_equal(out$match_method, "substring") - expect_equal(out$row$canonical_govid, "052037010") + expect_equal(out$row$canonical_govid, "062073207598") }) # ---- basket mode public surface ---- @@ -254,7 +258,7 @@ test_that("cog_gov_search basket mode resolves clean inputs in input order", { ) expect_s3_class(basket, "tbl_df") expect_equal(nrow(basket), 3L) - expect_equal(basket$canonical_govid, c("101006006", "052037010", "442227001")) + expect_equal(basket$canonical_govid, c("121011212191", "062073207598", "482453176394")) expect_equal(basket$gov_name, c("BROWARD COUNTY", "SAN DIEGO CITY", "AUSTIN CITY")) }) @@ -284,7 +288,7 @@ test_that("cog_gov_search basket mode skips ambiguous and no_match rows", { )) # Broward resolves; San Diego ambiguous; Notarealplace no_match. expect_equal(nrow(basket), 1L) - expect_equal(basket$canonical_govid, "101006006") + expect_equal(basket$canonical_govid, "121011212191") res <- attr(basket, "resolution") expect_equal(nrow(res), 3L) expect_equal(res$status, c("resolved", "ambiguous", "no_match")) @@ -306,20 +310,24 @@ test_that("cog_gov_search basket mode recycles single state", { state = "CA" ) expect_equal(nrow(basket), 2L) - expect_equal(basket$canonical_govid, c("052037010", "052001009")) + expect_equal(basket$canonical_govid, c("062073207598", "062001123093")) }) test_that("cog_gov_search basket mode within-type largest_pop records candidates", { skip_if_no_corpus() + # `type = "city"` for the Miami row pins the match set to govs_type = 2; + # under Phase P canonical naming MIAMI-DADE COUNTY also contains "Miami" + # and would otherwise make this an ambiguous (cross-type) match. basket <- suppressMessages(cog_gov_search( name = c("Miami", "OAKLAND CITY"), - state = c("FL", "CA") + state = c("FL", "CA"), + type = c("city", NA) )) expect_equal(nrow(basket), 2L) res <- attr(basket, "resolution") miami_row <- res[res$query_name == "Miami", ] expect_equal(miami_row$status, "largest_pop") - expect_equal(miami_row$canonical_govid, "102013013") + expect_equal(miami_row$canonical_govid, "122086194757") expect_gte(miami_row$n_candidates, 2L) expect_gte(nrow(miami_row$candidates[[1]]), 2L) }) @@ -396,7 +404,7 @@ test_that("cog_gov_search basket mode skips per-row excluded type without aborti )) # Broward should resolve; the special_district row should be no_match. expect_equal(nrow(basket), 1L) - expect_equal(basket$canonical_govid, "101006006") + expect_equal(basket$canonical_govid, "121011212191") res <- attr(basket, "resolution") expect_equal(res$status, c("resolved", "no_match")) # query_type should record what the user passed for the excluded-type row diff --git a/tests/testthat/test-spending.R b/tests/testthat/test-spending.R index 6f69faf..c767e6b 100644 --- a/tests/testthat/test-spending.R +++ b/tests/testthat/test-spending.R @@ -1,12 +1,12 @@ test_that("cog_spending returns expected shape for Broward Corrections 2020", { skip_if_no_corpus() - r <- cog_spending("101006006", years = 2020L, category = "Corrections") + r <- cog_spending("121011212191", years = 2020L, category = "Corrections") expect_s3_class(r, "tbl_df") expected_cols <- c("year", "canonical_govid", "gov_name", "spend_subtype", "category", "amt_nominal", "codes_included", "aggregate_fallback", "notes") expect_true(all(expected_cols %in% names(r))) - expect_equal(unique(r$canonical_govid), "101006006") + expect_equal(unique(r$canonical_govid), "121011212191") expect_equal(unique(r$year), 2020L) expect_equal(unique(r$category), "Corrections") expect_true(all(r$spend_subtype %in% c("operations", "capital"))) @@ -15,7 +15,7 @@ test_that("cog_spending returns expected shape for Broward Corrections 2020", { test_that("cog_spending vectorised years + categories", { skip_if_no_corpus() - r <- cog_spending("101006006", 2019:2020, + r <- cog_spending("121011212191", 2019:2020, category = c("Corrections", "Police")) expect_true(all(r$year %in% 2019:2020)) expect_true(all(r$category %in% c("Corrections", "Police"))) @@ -24,7 +24,7 @@ test_that("cog_spending vectorised years + categories", { test_that("cog_spending with per_capita adds per-capita nominal column", { skip_if_no_corpus() - r <- cog_spending("101006006", 2020L, "Corrections", per_capita = TRUE) + r <- cog_spending("121011212191", 2020L, "Corrections", per_capita = TRUE) expect_true("amt_per_capita_nominal" %in% names(r)) expect_false("amt_real" %in% names(r)) expect_false("amt_per_capita_real" %in% names(r)) @@ -34,7 +34,7 @@ test_that("cog_spending with per_capita adds per-capita nominal column", { test_that("cog_spending with adjust_to_year adds real column", { skip_if_no_corpus() - r <- cog_spending("101006006", 2019:2020, "Corrections", + r <- cog_spending("121011212191", 2019:2020, "Corrections", adjust_to_year = 2022L) expect_true("amt_real" %in% names(r)) r2019 <- dplyr::filter(r, year == 2019L) @@ -43,7 +43,7 @@ test_that("cog_spending with adjust_to_year adds real column", { test_that("cog_spending with per_capita + adjust_to_year adds all columns", { skip_if_no_corpus() - r <- cog_spending("101006006", 2020L, "Corrections", + r <- cog_spending("121011212191", 2020L, "Corrections", per_capita = TRUE, adjust_to_year = 2022L) expect_true(all(c("amt_nominal", "amt_real", "amt_per_capita_nominal", "amt_per_capita_real") %in% @@ -68,16 +68,16 @@ test_that("cog_spending for unknown govid returns empty tibble + informs", { test_that("cog_spending records found + missing govids in provenance", { skip_if_no_corpus() suppressMessages( - r <- cog_spending(c("101006006", "XXXINVALID"), 2020L, "Corrections") + r <- cog_spending(c("121011212191", "XXXINVALID"), 2020L, "Corrections") ) prov <- attr(r, "provenance") - expect_equal(sort(prov$scope$govids_found), "101006006") + expect_equal(sort(prov$scope$govids_found), "121011212191") expect_equal(sort(prov$scope$govids_missing), "XXXINVALID") }) test_that("cog_spending result has provenance attribute matching schema", { skip_if_no_corpus() - r <- cog_spending("101006006", 2020L, "Corrections") + r <- cog_spending("121011212191", 2020L, "Corrections") prov <- attr(r, "provenance") expect_type(prov, "list") expect_equal(prov$verb, "cog_spending") @@ -93,7 +93,7 @@ test_that("cog_spending result has provenance attribute matching schema", { test_that("cog_spending rejects invalid inputs", { expect_error(cog_spending(list(), 2020L), "character|data frame") - expect_error(cog_spending("101006006", "2020"), "years") + expect_error(cog_spending("121011212191", "2020"), "years") }) test_that("cog_spending accepts a cog_gov_search result directly", { @@ -101,12 +101,12 @@ test_that("cog_spending accepts a cog_gov_search result directly", { picks <- cog_gov_search("^BROWARD COUNTY$", state = "FL", type = "county") expect_gt(nrow(picks), 0L) r <- cog_spending(picks, 2020L, "Corrections") - expect_equal(unique(r$canonical_govid), "101006006") + expect_equal(unique(r$canonical_govid), "121011212191") }) test_that("cog_spending accepts a cog_find_peers result directly", { skip_if_no_corpus() - peers <- cog_find_peers("101006006", max_peers = 3L) + peers <- cog_find_peers("121011212191", max_peers = 3L) r <- cog_spending(peers, 2020L, "Police") expect_setequal(unique(r$canonical_govid), sort(peers$canonical_govid)) @@ -132,7 +132,7 @@ test_that("cog_spending accepts a basket-mode cog_gov_search result", { test_that("per_capita denominator is the per-year F-33 population", { skip_if_no_corpus() with_fixture_corpus({ - r <- cog_spending("101006006", years = 2019:2020, + r <- cog_spending("121011212191", years = 2019:2020, category = "Police", per_capita = TRUE) r_ops <- r[r$spend_subtype == "operations", ] # Implied denominator from amt_nominal / amt_per_capita_nominal @@ -140,9 +140,10 @@ test_that("per_capita denominator is the per-year F-33 population", { names(implied_pop) <- r_ops$year # Use absolute tolerance: within 1 person of per-year F-33 values. # Hardcoded values are Broward County's per-year Census F-33 population - # from the bundled fixture (regenerated 2026-04-29 against cog_pipeline - # aad34c6 + bd3e744). 1,940,907 is the static ACS 2018-2022 5-year value - # the legacy implementation would use; we assert it is NOT what we get. + # from the bundled fixture (regenerated 2026-07-11 against cog_pipeline + # publish tree, pipeline_commit 1a00925, Phase P schema_version 4). + # 1,940,907 is the static ACS 2018-2022 5-year value the legacy + # implementation would use; we assert it is NOT what we get. expect_true(abs(implied_pop[["2019"]] - 1935878) < 1) expect_true(abs(implied_pop[["2020"]] - 1952778) < 1) expect_false(all(abs(implied_pop - 1940907) < 1)) @@ -152,7 +153,7 @@ test_that("per_capita denominator is the per-year F-33 population", { test_that("pop_source = 'census_f33' does not produce unavailable-pop note", { skip_if_no_corpus() with_fixture_corpus({ - r <- cog_spending("101006006", years = 2019L, + r <- cog_spending("121011212191", years = 2019L, category = "Police", per_capita = TRUE) expect_true(all(r$pop_source == "census_f33")) expect_true(all(is.na(r$notes) | r$notes == "" | @@ -177,7 +178,7 @@ test_that("aggregate fallback + unavailable pop produce concatenated notes", { test_that("provenance records per-year denominator metadata", { skip_if_no_corpus() with_fixture_corpus({ - r <- cog_spending("101006006", years = 2019:2020, + r <- cog_spending("121011212191", years = 2019:2020, category = "Police", per_capita = TRUE) pc <- attr(r, "provenance")$transformations$per_capita expect_true(pc$applied) diff --git a/tests/testthat/test-views.R b/tests/testthat/test-views.R index 0d5b526..2d1255a 100644 --- a/tests/testthat/test-views.R +++ b/tests/testthat/test-views.R @@ -66,15 +66,16 @@ test_that("gov_population_yearly exposes one row per (year, canonical_govid)", { con, "SELECT year, canonical_govid, population, popyear FROM gov_population_yearly - WHERE canonical_govid = '101006006' + WHERE canonical_govid = '121011212191' ORDER BY year" ) expect_setequal(df$year, c(2019L, 2020L)) expect_equal(nrow(df), 2L) expect_true(all(!is.na(df$population))) - # Hardcoded values are from the bundled fixture (regenerated 2026-04-29 - # against cog_pipeline aad34c6 + bd3e744). Update if the fixture is - # rebuilt against a different source vintage. + # Hardcoded values are from the bundled fixture (regenerated 2026-07-11 + # against cog_pipeline publish tree, pipeline_commit 1a00925, Phase P + # schema_version 4). Update if the fixture is rebuilt against a + # different source vintage. expect_equal(df$population[df$year == 2019L], 1935878L) expect_equal(df$population[df$year == 2020L], 1952778L) # Uniqueness on (year, canonical_govid) across the whole view. diff --git a/vignettes/population-denominators.Rmd b/vignettes/population-denominators.Rmd index fae9596..0c65873 100644 --- a/vignettes/population-denominators.Rmd +++ b/vignettes/population-denominators.Rmd @@ -48,8 +48,8 @@ Census sometimes uses a population estimate from one year prior to the fiscal ye ```r years <- 2010:2023 out <- purrr::map_dfr(years, function(y) { - peers <- cog_find_peers("231082082", year = y, max_peers = 10L) - cog_peer_compare("231082082", peers, + peers <- cog_find_peers("261163166615", year = y, max_peers = 10L) + cog_peer_compare("261163166615", peers, category = "Police", years = y, per_capita = TRUE) })