feat: regenerate fixture corpus from Phase P publish tree + committed regen script
Adds data-raw/regenerate_fixture_corpus.R, parameterized by publish-cache path, so the fixture is never a manual rebuild again. Regenerates the 2019/2020 long partitions (byte-for-byte copy), the full 39,377-row canonical_fips_xwalk master, the new 117,503-row canonical_alias lookup table, and summary_categories from the Phase P publish tree; resyncs the four fixture docs; and hand-builds manifest.json with schema_version 4 and freshly computed sha256/row_count/size_bytes for every shipped file. Fixture grows from 3.7MB to 5.5MB, well under the 25MB budget.
This commit is contained in:
@@ -0,0 +1,215 @@
|
||||
# data-raw/regenerate_fixture_corpus.R
|
||||
#
|
||||
# Regenerate inst/extdata/fixture_corpus/ from a cog_pipeline publish tree.
|
||||
#
|
||||
# What this does:
|
||||
# 1. Copies the year=2019 and year=2020 long partitions as-is (byte-for-
|
||||
# byte) from <publish_cache>/data/long/ into the fixture.
|
||||
# 2. Copies the full canonical_fips_xwalk.parquet, canonical_alias.parquet,
|
||||
# and summary_categories.parquet metadata tables as-is (these are small
|
||||
# cross-vintage registries, not partitioned by year, so the fixture
|
||||
# ships the complete tables rather than a year-scoped subset).
|
||||
# 3. Resyncs the four reference docs (data_dictionary.md,
|
||||
# reader-specification.md, README.md, series_breaks.md) from the
|
||||
# publish tree's docs/.
|
||||
# 4. Hand-builds manifest.json for just the files the fixture ships,
|
||||
# following the shape of the previous fixture manifest but with
|
||||
# schema_version bumped to whatever the source manifest reports, and
|
||||
# freshly computed sha256 / row_count / size_bytes for every fixture
|
||||
# file (never copied from the source manifest, since paths and byte
|
||||
# layout can differ subtly between a full corpus and a fixture).
|
||||
#
|
||||
# This is never a manual job: run it whenever cog_pipeline publishes a new
|
||||
# corpus vintage that the fixture should track.
|
||||
#
|
||||
# Usage (from the uscogdata package root):
|
||||
# Rscript data-raw/regenerate_fixture_corpus.R
|
||||
# Rscript data-raw/regenerate_fixture_corpus.R /path/to/publish_cache
|
||||
#
|
||||
# Or from R:
|
||||
# source("data-raw/regenerate_fixture_corpus.R")
|
||||
# regenerate_fixture_corpus(publish_cache_dir = "/path/to/publish_cache")
|
||||
|
||||
regenerate_fixture_corpus <- function(
|
||||
publish_cache_dir = file.path(
|
||||
"..", "cog_pipeline", "_targets", "publish_cache"
|
||||
),
|
||||
fixture_dir = file.path("inst", "extdata", "fixture_corpus"),
|
||||
fixture_years = c(2019L, 2020L)) {
|
||||
stopifnot(
|
||||
requireNamespace("digest", quietly = TRUE),
|
||||
requireNamespace("jsonlite", quietly = TRUE),
|
||||
requireNamespace("duckdb", quietly = TRUE),
|
||||
requireNamespace("DBI", quietly = TRUE)
|
||||
)
|
||||
|
||||
publish_cache_dir <- normalizePath(publish_cache_dir, mustWork = TRUE)
|
||||
if (!dir.exists(fixture_dir)) dir.create(fixture_dir, recursive = TRUE)
|
||||
|
||||
source_manifest <- jsonlite::fromJSON(
|
||||
file.path(publish_cache_dir, "manifest.json"),
|
||||
simplifyVector = TRUE
|
||||
)
|
||||
|
||||
.copy_long_partitions(publish_cache_dir, fixture_dir, fixture_years)
|
||||
.copy_metadata_parquets(publish_cache_dir, fixture_dir)
|
||||
.copy_docs(publish_cache_dir, fixture_dir)
|
||||
|
||||
manifest <- .build_fixture_manifest(
|
||||
fixture_dir, source_manifest, fixture_years
|
||||
)
|
||||
manifest_path <- file.path(fixture_dir, "manifest.json")
|
||||
writeLines(
|
||||
jsonlite::toJSON(manifest, auto_unbox = TRUE, pretty = TRUE, null = "null"),
|
||||
manifest_path
|
||||
)
|
||||
|
||||
size_bytes <- sum(file.info(
|
||||
list.files(fixture_dir, recursive = TRUE, full.names = TRUE)
|
||||
)$size)
|
||||
message(sprintf(
|
||||
"Fixture corpus regenerated at %s (%.2f MB total).",
|
||||
fixture_dir, size_bytes / 1024^2
|
||||
))
|
||||
invisible(manifest)
|
||||
}
|
||||
|
||||
# Copy each requested year's partition directory (just the parquet file
|
||||
# inside it) from the publish tree into the fixture, as-is.
|
||||
#' @noRd
|
||||
.copy_long_partitions <- function(publish_cache_dir, fixture_dir, years) {
|
||||
for (yr in years) {
|
||||
part_rel <- file.path("data", "long", sprintf("year=%d", yr), "part-0.parquet")
|
||||
src <- file.path(publish_cache_dir, part_rel)
|
||||
dst <- file.path(fixture_dir, part_rel)
|
||||
if (!file.exists(src)) {
|
||||
stop(sprintf("Source partition missing: %s", src))
|
||||
}
|
||||
dir.create(dirname(dst), recursive = TRUE, showWarnings = FALSE)
|
||||
ok <- file.copy(src, dst, overwrite = TRUE)
|
||||
if (!ok) stop(sprintf("Failed to copy %s -> %s", src, dst))
|
||||
}
|
||||
invisible(NULL)
|
||||
}
|
||||
|
||||
# Copy the full (not year-scoped) canonical_fips_xwalk, canonical_alias, and
|
||||
# summary_categories parquet tables.
|
||||
#' @noRd
|
||||
.copy_metadata_parquets <- function(publish_cache_dir, fixture_dir) {
|
||||
files <- c(
|
||||
"canonical_fips_xwalk.parquet",
|
||||
"canonical_alias.parquet",
|
||||
"summary_categories.parquet"
|
||||
)
|
||||
for (f in files) {
|
||||
src <- file.path(publish_cache_dir, "data", f)
|
||||
dst <- file.path(fixture_dir, "data", f)
|
||||
if (!file.exists(src)) {
|
||||
stop(sprintf("Source metadata file missing: %s", src))
|
||||
}
|
||||
dir.create(dirname(dst), recursive = TRUE, showWarnings = FALSE)
|
||||
ok <- file.copy(src, dst, overwrite = TRUE)
|
||||
if (!ok) stop(sprintf("Failed to copy %s -> %s", src, dst))
|
||||
}
|
||||
invisible(NULL)
|
||||
}
|
||||
|
||||
# Resync the four reference docs shipped alongside the fixture.
|
||||
#' @noRd
|
||||
.copy_docs <- function(publish_cache_dir, fixture_dir) {
|
||||
docs <- c(
|
||||
"data_dictionary.md", "reader-specification.md",
|
||||
"README.md", "series_breaks.md"
|
||||
)
|
||||
dst_dir <- file.path(fixture_dir, "docs")
|
||||
dir.create(dst_dir, recursive = TRUE, showWarnings = FALSE)
|
||||
for (f in docs) {
|
||||
src <- file.path(publish_cache_dir, "docs", f)
|
||||
if (!file.exists(src)) {
|
||||
stop(sprintf("Source doc missing: %s", src))
|
||||
}
|
||||
ok <- file.copy(src, file.path(dst_dir, f), overwrite = TRUE)
|
||||
if (!ok) stop(sprintf("Failed to copy doc %s", f))
|
||||
}
|
||||
invisible(NULL)
|
||||
}
|
||||
|
||||
# Count rows in a parquet file via an ephemeral DuckDB connection.
|
||||
#' @noRd
|
||||
.parquet_row_count <- function(path) {
|
||||
con <- DBI::dbConnect(duckdb::duckdb())
|
||||
on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE)
|
||||
DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT COUNT(*) AS n FROM read_parquet(%s)",
|
||||
.sql_quote(path)
|
||||
))$n
|
||||
}
|
||||
|
||||
#' @noRd
|
||||
.sql_quote <- function(x) paste0("'", gsub("'", "''", x), "'")
|
||||
|
||||
# Hand-build manifest.json following the shape of the previous fixture
|
||||
# manifest: schema_version / built_at / pipeline_commit / fixture_note /
|
||||
# data_vintage / scope / schema / files.long_partitions / files.metadata /
|
||||
# series_breaks_ref / reader_spec_ref. Every sha256 / row_count / size_bytes
|
||||
# is freshly computed against the files actually written into fixture_dir.
|
||||
#' @noRd
|
||||
.build_fixture_manifest <- function(fixture_dir, source_manifest, years) {
|
||||
long_partitions <- lapply(years, function(yr) {
|
||||
rel <- file.path("data", "long", sprintf("year=%d", yr), "part-0.parquet")
|
||||
path <- file.path(fixture_dir, rel)
|
||||
list(
|
||||
year = as.integer(yr),
|
||||
path = gsub("\\\\", "/", rel),
|
||||
sha256 = digest::digest(path, algo = "sha256", file = TRUE),
|
||||
row_count = as.integer(.parquet_row_count(path)),
|
||||
size_bytes = as.integer(file.info(path)$size)
|
||||
)
|
||||
})
|
||||
|
||||
metadata_files <- c(
|
||||
"canonical_alias.parquet",
|
||||
"canonical_fips_xwalk.parquet",
|
||||
"summary_categories.parquet"
|
||||
)
|
||||
metadata <- lapply(metadata_files, function(f) {
|
||||
rel <- file.path("data", f)
|
||||
path <- file.path(fixture_dir, rel)
|
||||
list(
|
||||
path = gsub("\\\\", "/", rel),
|
||||
sha256 = digest::digest(path, algo = "sha256", file = TRUE),
|
||||
description = f
|
||||
)
|
||||
})
|
||||
|
||||
list(
|
||||
schema_version = as.integer(source_manifest$schema_version),
|
||||
built_at = format(Sys.time(), "%Y-%m-%dT%H:%M:%SZ", tz = "UTC"),
|
||||
pipeline_commit = source_manifest$pipeline_commit,
|
||||
fixture_note = paste(
|
||||
"Two-year (2019-2020) fixture for uscogdata tests. Full corpus",
|
||||
"available via USCOGDATA_URL. Regenerated for Phase P",
|
||||
"(schema_version 4, uniformly 12-char canonical_govid) with the full",
|
||||
"canonical_fips_xwalk master and the new canonical_alias lookup",
|
||||
"table via data-raw/regenerate_fixture_corpus.R."
|
||||
),
|
||||
data_vintage = source_manifest$data_vintage,
|
||||
scope = source_manifest$scope,
|
||||
schema = source_manifest$schema,
|
||||
files = list(
|
||||
long_partitions = long_partitions,
|
||||
metadata = metadata
|
||||
),
|
||||
series_breaks_ref = source_manifest$series_breaks_ref,
|
||||
reader_spec_ref = source_manifest$reader_spec_ref
|
||||
)
|
||||
}
|
||||
|
||||
if (identical(environment(), globalenv()) && sys.nframe() == 0L) {
|
||||
args <- commandArgs(trailingOnly = TRUE)
|
||||
if (length(args) >= 1L) {
|
||||
regenerate_fixture_corpus(publish_cache_dir = args[[1]])
|
||||
} else {
|
||||
regenerate_fixture_corpus()
|
||||
}
|
||||
}
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+18
-46
@@ -1,54 +1,21 @@
|
||||
{
|
||||
"schema_version": 3,
|
||||
"built_at": "2026-04-29T18:57:18Z",
|
||||
"pipeline_commit": "bd3e744",
|
||||
"fixture_note": "Two-year (2019-2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated with Layer 1 (canonical_govid resolver gate) + Layer 2 (PID-era extended xwalk).",
|
||||
"schema_version": 4,
|
||||
"built_at": "2026-07-11T13:24:01Z",
|
||||
"pipeline_commit": "1a00925",
|
||||
"fixture_note": "Two-year (2019-2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated for Phase P (schema_version 4, uniformly 12-char canonical_govid) with the full canonical_fips_xwalk master and the new canonical_alias lookup table via data-raw/regenerate_fixture_corpus.R.",
|
||||
"data_vintage": {
|
||||
"census_source_downloaded": "unknown",
|
||||
"cpi_vintage": "FRED CPIAUCSL",
|
||||
"acs_vintage": "ACS 2018-2022 5-year"
|
||||
},
|
||||
"scope": {
|
||||
"gov_types_included": [
|
||||
0,
|
||||
1,
|
||||
2,
|
||||
3
|
||||
],
|
||||
"gov_types_excluded": [
|
||||
4,
|
||||
5
|
||||
],
|
||||
"gov_types_included": [0, 1, 2, 3],
|
||||
"gov_types_excluded": [4, 5],
|
||||
"scope_note": "v0.1 covers state, county, city/municipality, and township governments. Special districts (type 4) and school districts (type 5) are excluded pending validation in a future cycle."
|
||||
},
|
||||
"schema": {
|
||||
"long_column_count": 24,
|
||||
"long_columns": [
|
||||
"fips_state",
|
||||
"type",
|
||||
"fips_county",
|
||||
"govid",
|
||||
"gov_blank",
|
||||
"gov_name",
|
||||
"county_name",
|
||||
"fips_state_code",
|
||||
"fips_county_code",
|
||||
"fips_place_code",
|
||||
"population",
|
||||
"popyear",
|
||||
"enrollment",
|
||||
"enrollyear",
|
||||
"function_code",
|
||||
"sch_level_code",
|
||||
"fiscal_year_end",
|
||||
"srvy_year",
|
||||
"item_code",
|
||||
"amt",
|
||||
"srv_data",
|
||||
"impute_flag",
|
||||
"is_aggregate",
|
||||
"canonical_govid"
|
||||
],
|
||||
"long_columns": ["fips_state", "type", "fips_county", "govid", "gov_blank", "gov_name", "county_name", "fips_state_code", "fips_county_code", "fips_place_code", "population", "popyear", "enrollment", "enrollyear", "function_code", "sch_level_code", "fiscal_year_end", "srvy_year", "item_code", "amt", "srv_data", "impute_flag", "is_aggregate", "canonical_govid"],
|
||||
"data_dictionary": "docs/data_dictionary.md"
|
||||
},
|
||||
"files": {
|
||||
@@ -56,27 +23,32 @@
|
||||
{
|
||||
"year": 2019,
|
||||
"path": "data/long/year=2019/part-0.parquet",
|
||||
"sha256": "d93affbbf9c46bd193fa5e07d89dfa9b08ad4f6fc2442fe7a27e6e4ef28f8bd9",
|
||||
"sha256": "c0a2bf0758af129d5dfddb6ff6665cc435ddee87fd6879e788fb56ed53ab22b8",
|
||||
"row_count": 318139,
|
||||
"size_bytes": 1425125
|
||||
"size_bytes": 1441404
|
||||
},
|
||||
{
|
||||
"year": 2020,
|
||||
"path": "data/long/year=2020/part-0.parquet",
|
||||
"sha256": "c9224833bd914cc85efb2ec62f83f75c34dfa469d4398f5cb1789fd028c03389",
|
||||
"sha256": "92570b9d55ec3425d034db37838f91c3b8359d0454d3d98730a6016b62e4bb48",
|
||||
"row_count": 317500,
|
||||
"size_bytes": 1428049
|
||||
"size_bytes": 1444011
|
||||
}
|
||||
],
|
||||
"metadata": [
|
||||
{
|
||||
"path": "data/canonical_alias.parquet",
|
||||
"sha256": "db784d4ec9e5abb4b033405627d8bdd6f8d8ff96b3033a5b2587b4b59113ec9c",
|
||||
"description": "canonical_alias.parquet"
|
||||
},
|
||||
{
|
||||
"path": "data/canonical_fips_xwalk.parquet",
|
||||
"sha256": "4bbdf0415b0ae1a24927869bb9e8e75eca5d7bcceebdf8c26346c7cbf8472d34",
|
||||
"sha256": "c0b1295e779b601f2782de40ab143556324e74d30d214fc10997c2244c109389",
|
||||
"description": "canonical_fips_xwalk.parquet"
|
||||
},
|
||||
{
|
||||
"path": "data/summary_categories.parquet",
|
||||
"sha256": "60045e22bc2723318fa2cb73f8e5038250dc54d24b3447c6750dfe29035335b8",
|
||||
"sha256": "dd59e7f58a022679ad43511c8c8e938b8dd4be81196bbeeee21e67bdcca2295b",
|
||||
"description": "summary_categories.parquet"
|
||||
}
|
||||
]
|
||||
|
||||
Reference in New Issue
Block a user