diff --git a/data-raw/regenerate_fixture_corpus.R b/data-raw/regenerate_fixture_corpus.R new file mode 100644 index 0000000..9e54756 --- /dev/null +++ b/data-raw/regenerate_fixture_corpus.R @@ -0,0 +1,215 @@ +# data-raw/regenerate_fixture_corpus.R +# +# Regenerate inst/extdata/fixture_corpus/ from a cog_pipeline publish tree. +# +# What this does: +# 1. Copies the year=2019 and year=2020 long partitions as-is (byte-for- +# byte) from /data/long/ into the fixture. +# 2. Copies the full canonical_fips_xwalk.parquet, canonical_alias.parquet, +# and summary_categories.parquet metadata tables as-is (these are small +# cross-vintage registries, not partitioned by year, so the fixture +# ships the complete tables rather than a year-scoped subset). +# 3. Resyncs the four reference docs (data_dictionary.md, +# reader-specification.md, README.md, series_breaks.md) from the +# publish tree's docs/. +# 4. Hand-builds manifest.json for just the files the fixture ships, +# following the shape of the previous fixture manifest but with +# schema_version bumped to whatever the source manifest reports, and +# freshly computed sha256 / row_count / size_bytes for every fixture +# file (never copied from the source manifest, since paths and byte +# layout can differ subtly between a full corpus and a fixture). +# +# This is never a manual job: run it whenever cog_pipeline publishes a new +# corpus vintage that the fixture should track. +# +# Usage (from the uscogdata package root): +# Rscript data-raw/regenerate_fixture_corpus.R +# Rscript data-raw/regenerate_fixture_corpus.R /path/to/publish_cache +# +# Or from R: +# source("data-raw/regenerate_fixture_corpus.R") +# regenerate_fixture_corpus(publish_cache_dir = "/path/to/publish_cache") + +regenerate_fixture_corpus <- function( + publish_cache_dir = file.path( + "..", "cog_pipeline", "_targets", "publish_cache" + ), + fixture_dir = file.path("inst", "extdata", "fixture_corpus"), + fixture_years = c(2019L, 2020L)) { + stopifnot( + requireNamespace("digest", quietly = TRUE), + requireNamespace("jsonlite", quietly = TRUE), + requireNamespace("duckdb", quietly = TRUE), + requireNamespace("DBI", quietly = TRUE) + ) + + publish_cache_dir <- normalizePath(publish_cache_dir, mustWork = TRUE) + if (!dir.exists(fixture_dir)) dir.create(fixture_dir, recursive = TRUE) + + source_manifest <- jsonlite::fromJSON( + file.path(publish_cache_dir, "manifest.json"), + simplifyVector = TRUE + ) + + .copy_long_partitions(publish_cache_dir, fixture_dir, fixture_years) + .copy_metadata_parquets(publish_cache_dir, fixture_dir) + .copy_docs(publish_cache_dir, fixture_dir) + + manifest <- .build_fixture_manifest( + fixture_dir, source_manifest, fixture_years + ) + manifest_path <- file.path(fixture_dir, "manifest.json") + writeLines( + jsonlite::toJSON(manifest, auto_unbox = TRUE, pretty = TRUE, null = "null"), + manifest_path + ) + + size_bytes <- sum(file.info( + list.files(fixture_dir, recursive = TRUE, full.names = TRUE) + )$size) + message(sprintf( + "Fixture corpus regenerated at %s (%.2f MB total).", + fixture_dir, size_bytes / 1024^2 + )) + invisible(manifest) +} + +# Copy each requested year's partition directory (just the parquet file +# inside it) from the publish tree into the fixture, as-is. +#' @noRd +.copy_long_partitions <- function(publish_cache_dir, fixture_dir, years) { + for (yr in years) { + part_rel <- file.path("data", "long", sprintf("year=%d", yr), "part-0.parquet") + src <- file.path(publish_cache_dir, part_rel) + dst <- file.path(fixture_dir, part_rel) + if (!file.exists(src)) { + stop(sprintf("Source partition missing: %s", src)) + } + dir.create(dirname(dst), recursive = TRUE, showWarnings = FALSE) + ok <- file.copy(src, dst, overwrite = TRUE) + if (!ok) stop(sprintf("Failed to copy %s -> %s", src, dst)) + } + invisible(NULL) +} + +# Copy the full (not year-scoped) canonical_fips_xwalk, canonical_alias, and +# summary_categories parquet tables. +#' @noRd +.copy_metadata_parquets <- function(publish_cache_dir, fixture_dir) { + files <- c( + "canonical_fips_xwalk.parquet", + "canonical_alias.parquet", + "summary_categories.parquet" + ) + for (f in files) { + src <- file.path(publish_cache_dir, "data", f) + dst <- file.path(fixture_dir, "data", f) + if (!file.exists(src)) { + stop(sprintf("Source metadata file missing: %s", src)) + } + dir.create(dirname(dst), recursive = TRUE, showWarnings = FALSE) + ok <- file.copy(src, dst, overwrite = TRUE) + if (!ok) stop(sprintf("Failed to copy %s -> %s", src, dst)) + } + invisible(NULL) +} + +# Resync the four reference docs shipped alongside the fixture. +#' @noRd +.copy_docs <- function(publish_cache_dir, fixture_dir) { + docs <- c( + "data_dictionary.md", "reader-specification.md", + "README.md", "series_breaks.md" + ) + dst_dir <- file.path(fixture_dir, "docs") + dir.create(dst_dir, recursive = TRUE, showWarnings = FALSE) + for (f in docs) { + src <- file.path(publish_cache_dir, "docs", f) + if (!file.exists(src)) { + stop(sprintf("Source doc missing: %s", src)) + } + ok <- file.copy(src, file.path(dst_dir, f), overwrite = TRUE) + if (!ok) stop(sprintf("Failed to copy doc %s", f)) + } + invisible(NULL) +} + +# Count rows in a parquet file via an ephemeral DuckDB connection. +#' @noRd +.parquet_row_count <- function(path) { + con <- DBI::dbConnect(duckdb::duckdb()) + on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE) + DBI::dbGetQuery(con, sprintf( + "SELECT COUNT(*) AS n FROM read_parquet(%s)", + .sql_quote(path) + ))$n +} + +#' @noRd +.sql_quote <- function(x) paste0("'", gsub("'", "''", x), "'") + +# Hand-build manifest.json following the shape of the previous fixture +# manifest: schema_version / built_at / pipeline_commit / fixture_note / +# data_vintage / scope / schema / files.long_partitions / files.metadata / +# series_breaks_ref / reader_spec_ref. Every sha256 / row_count / size_bytes +# is freshly computed against the files actually written into fixture_dir. +#' @noRd +.build_fixture_manifest <- function(fixture_dir, source_manifest, years) { + long_partitions <- lapply(years, function(yr) { + rel <- file.path("data", "long", sprintf("year=%d", yr), "part-0.parquet") + path <- file.path(fixture_dir, rel) + list( + year = as.integer(yr), + path = gsub("\\\\", "/", rel), + sha256 = digest::digest(path, algo = "sha256", file = TRUE), + row_count = as.integer(.parquet_row_count(path)), + size_bytes = as.integer(file.info(path)$size) + ) + }) + + metadata_files <- c( + "canonical_alias.parquet", + "canonical_fips_xwalk.parquet", + "summary_categories.parquet" + ) + metadata <- lapply(metadata_files, function(f) { + rel <- file.path("data", f) + path <- file.path(fixture_dir, rel) + list( + path = gsub("\\\\", "/", rel), + sha256 = digest::digest(path, algo = "sha256", file = TRUE), + description = f + ) + }) + + list( + schema_version = as.integer(source_manifest$schema_version), + built_at = format(Sys.time(), "%Y-%m-%dT%H:%M:%SZ", tz = "UTC"), + pipeline_commit = source_manifest$pipeline_commit, + fixture_note = paste( + "Two-year (2019-2020) fixture for uscogdata tests. Full corpus", + "available via USCOGDATA_URL. Regenerated for Phase P", + "(schema_version 4, uniformly 12-char canonical_govid) with the full", + "canonical_fips_xwalk master and the new canonical_alias lookup", + "table via data-raw/regenerate_fixture_corpus.R." + ), + data_vintage = source_manifest$data_vintage, + scope = source_manifest$scope, + schema = source_manifest$schema, + files = list( + long_partitions = long_partitions, + metadata = metadata + ), + series_breaks_ref = source_manifest$series_breaks_ref, + reader_spec_ref = source_manifest$reader_spec_ref + ) +} + +if (identical(environment(), globalenv()) && sys.nframe() == 0L) { + args <- commandArgs(trailingOnly = TRUE) + if (length(args) >= 1L) { + regenerate_fixture_corpus(publish_cache_dir = args[[1]]) + } else { + regenerate_fixture_corpus() + } +} diff --git a/inst/extdata/fixture_corpus/data/canonical_alias.parquet b/inst/extdata/fixture_corpus/data/canonical_alias.parquet new file mode 100644 index 0000000..6115b7f Binary files /dev/null and b/inst/extdata/fixture_corpus/data/canonical_alias.parquet differ diff --git a/inst/extdata/fixture_corpus/data/canonical_fips_xwalk.parquet b/inst/extdata/fixture_corpus/data/canonical_fips_xwalk.parquet index f7b8868..fe2aed1 100644 Binary files a/inst/extdata/fixture_corpus/data/canonical_fips_xwalk.parquet and b/inst/extdata/fixture_corpus/data/canonical_fips_xwalk.parquet differ diff --git a/inst/extdata/fixture_corpus/data/long/year=2019/part-0.parquet b/inst/extdata/fixture_corpus/data/long/year=2019/part-0.parquet index 5e10a35..127794c 100644 Binary files a/inst/extdata/fixture_corpus/data/long/year=2019/part-0.parquet and b/inst/extdata/fixture_corpus/data/long/year=2019/part-0.parquet differ diff --git a/inst/extdata/fixture_corpus/data/long/year=2020/part-0.parquet b/inst/extdata/fixture_corpus/data/long/year=2020/part-0.parquet index f0f8075..4129531 100644 Binary files a/inst/extdata/fixture_corpus/data/long/year=2020/part-0.parquet and b/inst/extdata/fixture_corpus/data/long/year=2020/part-0.parquet differ diff --git a/inst/extdata/fixture_corpus/data/summary_categories.parquet b/inst/extdata/fixture_corpus/data/summary_categories.parquet index 8003ca4..b58d39d 100644 Binary files a/inst/extdata/fixture_corpus/data/summary_categories.parquet and b/inst/extdata/fixture_corpus/data/summary_categories.parquet differ diff --git a/inst/extdata/fixture_corpus/manifest.json b/inst/extdata/fixture_corpus/manifest.json index 600d1b0..9b87d9b 100644 --- a/inst/extdata/fixture_corpus/manifest.json +++ b/inst/extdata/fixture_corpus/manifest.json @@ -1,54 +1,21 @@ { - "schema_version": 3, - "built_at": "2026-04-29T18:57:18Z", - "pipeline_commit": "bd3e744", - "fixture_note": "Two-year (2019-2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated with Layer 1 (canonical_govid resolver gate) + Layer 2 (PID-era extended xwalk).", + "schema_version": 4, + "built_at": "2026-07-11T13:24:01Z", + "pipeline_commit": "1a00925", + "fixture_note": "Two-year (2019-2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated for Phase P (schema_version 4, uniformly 12-char canonical_govid) with the full canonical_fips_xwalk master and the new canonical_alias lookup table via data-raw/regenerate_fixture_corpus.R.", "data_vintage": { "census_source_downloaded": "unknown", "cpi_vintage": "FRED CPIAUCSL", "acs_vintage": "ACS 2018-2022 5-year" }, "scope": { - "gov_types_included": [ - 0, - 1, - 2, - 3 - ], - "gov_types_excluded": [ - 4, - 5 - ], + "gov_types_included": [0, 1, 2, 3], + "gov_types_excluded": [4, 5], "scope_note": "v0.1 covers state, county, city/municipality, and township governments. Special districts (type 4) and school districts (type 5) are excluded pending validation in a future cycle." }, "schema": { "long_column_count": 24, - "long_columns": [ - "fips_state", - "type", - "fips_county", - "govid", - "gov_blank", - "gov_name", - "county_name", - "fips_state_code", - "fips_county_code", - "fips_place_code", - "population", - "popyear", - "enrollment", - "enrollyear", - "function_code", - "sch_level_code", - "fiscal_year_end", - "srvy_year", - "item_code", - "amt", - "srv_data", - "impute_flag", - "is_aggregate", - "canonical_govid" - ], + "long_columns": ["fips_state", "type", "fips_county", "govid", "gov_blank", "gov_name", "county_name", "fips_state_code", "fips_county_code", "fips_place_code", "population", "popyear", "enrollment", "enrollyear", "function_code", "sch_level_code", "fiscal_year_end", "srvy_year", "item_code", "amt", "srv_data", "impute_flag", "is_aggregate", "canonical_govid"], "data_dictionary": "docs/data_dictionary.md" }, "files": { @@ -56,27 +23,32 @@ { "year": 2019, "path": "data/long/year=2019/part-0.parquet", - "sha256": "d93affbbf9c46bd193fa5e07d89dfa9b08ad4f6fc2442fe7a27e6e4ef28f8bd9", + "sha256": "c0a2bf0758af129d5dfddb6ff6665cc435ddee87fd6879e788fb56ed53ab22b8", "row_count": 318139, - "size_bytes": 1425125 + "size_bytes": 1441404 }, { "year": 2020, "path": "data/long/year=2020/part-0.parquet", - "sha256": "c9224833bd914cc85efb2ec62f83f75c34dfa469d4398f5cb1789fd028c03389", + "sha256": "92570b9d55ec3425d034db37838f91c3b8359d0454d3d98730a6016b62e4bb48", "row_count": 317500, - "size_bytes": 1428049 + "size_bytes": 1444011 } ], "metadata": [ + { + "path": "data/canonical_alias.parquet", + "sha256": "db784d4ec9e5abb4b033405627d8bdd6f8d8ff96b3033a5b2587b4b59113ec9c", + "description": "canonical_alias.parquet" + }, { "path": "data/canonical_fips_xwalk.parquet", - "sha256": "4bbdf0415b0ae1a24927869bb9e8e75eca5d7bcceebdf8c26346c7cbf8472d34", + "sha256": "c0b1295e779b601f2782de40ab143556324e74d30d214fc10997c2244c109389", "description": "canonical_fips_xwalk.parquet" }, { "path": "data/summary_categories.parquet", - "sha256": "60045e22bc2723318fa2cb73f8e5038250dc54d24b3447c6750dfe29035335b8", + "sha256": "dd59e7f58a022679ad43511c8c8e938b8dd4be81196bbeeee21e67bdcca2295b", "description": "summary_categories.parquet" } ]