feat: regenerate fixture corpus from Phase P publish tree + committed regen script
Adds data-raw/regenerate_fixture_corpus.R, parameterized by publish-cache path, so the fixture is never a manual rebuild again. Regenerates the 2019/2020 long partitions (byte-for-byte copy), the full 39,377-row canonical_fips_xwalk master, the new 117,503-row canonical_alias lookup table, and summary_categories from the Phase P publish tree; resyncs the four fixture docs; and hand-builds manifest.json with schema_version 4 and freshly computed sha256/row_count/size_bytes for every shipped file. Fixture grows from 3.7MB to 5.5MB, well under the 25MB budget.
This commit is contained in:
@@ -0,0 +1,215 @@
|
|||||||
|
# data-raw/regenerate_fixture_corpus.R
|
||||||
|
#
|
||||||
|
# Regenerate inst/extdata/fixture_corpus/ from a cog_pipeline publish tree.
|
||||||
|
#
|
||||||
|
# What this does:
|
||||||
|
# 1. Copies the year=2019 and year=2020 long partitions as-is (byte-for-
|
||||||
|
# byte) from <publish_cache>/data/long/ into the fixture.
|
||||||
|
# 2. Copies the full canonical_fips_xwalk.parquet, canonical_alias.parquet,
|
||||||
|
# and summary_categories.parquet metadata tables as-is (these are small
|
||||||
|
# cross-vintage registries, not partitioned by year, so the fixture
|
||||||
|
# ships the complete tables rather than a year-scoped subset).
|
||||||
|
# 3. Resyncs the four reference docs (data_dictionary.md,
|
||||||
|
# reader-specification.md, README.md, series_breaks.md) from the
|
||||||
|
# publish tree's docs/.
|
||||||
|
# 4. Hand-builds manifest.json for just the files the fixture ships,
|
||||||
|
# following the shape of the previous fixture manifest but with
|
||||||
|
# schema_version bumped to whatever the source manifest reports, and
|
||||||
|
# freshly computed sha256 / row_count / size_bytes for every fixture
|
||||||
|
# file (never copied from the source manifest, since paths and byte
|
||||||
|
# layout can differ subtly between a full corpus and a fixture).
|
||||||
|
#
|
||||||
|
# This is never a manual job: run it whenever cog_pipeline publishes a new
|
||||||
|
# corpus vintage that the fixture should track.
|
||||||
|
#
|
||||||
|
# Usage (from the uscogdata package root):
|
||||||
|
# Rscript data-raw/regenerate_fixture_corpus.R
|
||||||
|
# Rscript data-raw/regenerate_fixture_corpus.R /path/to/publish_cache
|
||||||
|
#
|
||||||
|
# Or from R:
|
||||||
|
# source("data-raw/regenerate_fixture_corpus.R")
|
||||||
|
# regenerate_fixture_corpus(publish_cache_dir = "/path/to/publish_cache")
|
||||||
|
|
||||||
|
regenerate_fixture_corpus <- function(
|
||||||
|
publish_cache_dir = file.path(
|
||||||
|
"..", "cog_pipeline", "_targets", "publish_cache"
|
||||||
|
),
|
||||||
|
fixture_dir = file.path("inst", "extdata", "fixture_corpus"),
|
||||||
|
fixture_years = c(2019L, 2020L)) {
|
||||||
|
stopifnot(
|
||||||
|
requireNamespace("digest", quietly = TRUE),
|
||||||
|
requireNamespace("jsonlite", quietly = TRUE),
|
||||||
|
requireNamespace("duckdb", quietly = TRUE),
|
||||||
|
requireNamespace("DBI", quietly = TRUE)
|
||||||
|
)
|
||||||
|
|
||||||
|
publish_cache_dir <- normalizePath(publish_cache_dir, mustWork = TRUE)
|
||||||
|
if (!dir.exists(fixture_dir)) dir.create(fixture_dir, recursive = TRUE)
|
||||||
|
|
||||||
|
source_manifest <- jsonlite::fromJSON(
|
||||||
|
file.path(publish_cache_dir, "manifest.json"),
|
||||||
|
simplifyVector = TRUE
|
||||||
|
)
|
||||||
|
|
||||||
|
.copy_long_partitions(publish_cache_dir, fixture_dir, fixture_years)
|
||||||
|
.copy_metadata_parquets(publish_cache_dir, fixture_dir)
|
||||||
|
.copy_docs(publish_cache_dir, fixture_dir)
|
||||||
|
|
||||||
|
manifest <- .build_fixture_manifest(
|
||||||
|
fixture_dir, source_manifest, fixture_years
|
||||||
|
)
|
||||||
|
manifest_path <- file.path(fixture_dir, "manifest.json")
|
||||||
|
writeLines(
|
||||||
|
jsonlite::toJSON(manifest, auto_unbox = TRUE, pretty = TRUE, null = "null"),
|
||||||
|
manifest_path
|
||||||
|
)
|
||||||
|
|
||||||
|
size_bytes <- sum(file.info(
|
||||||
|
list.files(fixture_dir, recursive = TRUE, full.names = TRUE)
|
||||||
|
)$size)
|
||||||
|
message(sprintf(
|
||||||
|
"Fixture corpus regenerated at %s (%.2f MB total).",
|
||||||
|
fixture_dir, size_bytes / 1024^2
|
||||||
|
))
|
||||||
|
invisible(manifest)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Copy each requested year's partition directory (just the parquet file
|
||||||
|
# inside it) from the publish tree into the fixture, as-is.
|
||||||
|
#' @noRd
|
||||||
|
.copy_long_partitions <- function(publish_cache_dir, fixture_dir, years) {
|
||||||
|
for (yr in years) {
|
||||||
|
part_rel <- file.path("data", "long", sprintf("year=%d", yr), "part-0.parquet")
|
||||||
|
src <- file.path(publish_cache_dir, part_rel)
|
||||||
|
dst <- file.path(fixture_dir, part_rel)
|
||||||
|
if (!file.exists(src)) {
|
||||||
|
stop(sprintf("Source partition missing: %s", src))
|
||||||
|
}
|
||||||
|
dir.create(dirname(dst), recursive = TRUE, showWarnings = FALSE)
|
||||||
|
ok <- file.copy(src, dst, overwrite = TRUE)
|
||||||
|
if (!ok) stop(sprintf("Failed to copy %s -> %s", src, dst))
|
||||||
|
}
|
||||||
|
invisible(NULL)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Copy the full (not year-scoped) canonical_fips_xwalk, canonical_alias, and
|
||||||
|
# summary_categories parquet tables.
|
||||||
|
#' @noRd
|
||||||
|
.copy_metadata_parquets <- function(publish_cache_dir, fixture_dir) {
|
||||||
|
files <- c(
|
||||||
|
"canonical_fips_xwalk.parquet",
|
||||||
|
"canonical_alias.parquet",
|
||||||
|
"summary_categories.parquet"
|
||||||
|
)
|
||||||
|
for (f in files) {
|
||||||
|
src <- file.path(publish_cache_dir, "data", f)
|
||||||
|
dst <- file.path(fixture_dir, "data", f)
|
||||||
|
if (!file.exists(src)) {
|
||||||
|
stop(sprintf("Source metadata file missing: %s", src))
|
||||||
|
}
|
||||||
|
dir.create(dirname(dst), recursive = TRUE, showWarnings = FALSE)
|
||||||
|
ok <- file.copy(src, dst, overwrite = TRUE)
|
||||||
|
if (!ok) stop(sprintf("Failed to copy %s -> %s", src, dst))
|
||||||
|
}
|
||||||
|
invisible(NULL)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Resync the four reference docs shipped alongside the fixture.
|
||||||
|
#' @noRd
|
||||||
|
.copy_docs <- function(publish_cache_dir, fixture_dir) {
|
||||||
|
docs <- c(
|
||||||
|
"data_dictionary.md", "reader-specification.md",
|
||||||
|
"README.md", "series_breaks.md"
|
||||||
|
)
|
||||||
|
dst_dir <- file.path(fixture_dir, "docs")
|
||||||
|
dir.create(dst_dir, recursive = TRUE, showWarnings = FALSE)
|
||||||
|
for (f in docs) {
|
||||||
|
src <- file.path(publish_cache_dir, "docs", f)
|
||||||
|
if (!file.exists(src)) {
|
||||||
|
stop(sprintf("Source doc missing: %s", src))
|
||||||
|
}
|
||||||
|
ok <- file.copy(src, file.path(dst_dir, f), overwrite = TRUE)
|
||||||
|
if (!ok) stop(sprintf("Failed to copy doc %s", f))
|
||||||
|
}
|
||||||
|
invisible(NULL)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Count rows in a parquet file via an ephemeral DuckDB connection.
|
||||||
|
#' @noRd
|
||||||
|
.parquet_row_count <- function(path) {
|
||||||
|
con <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE)
|
||||||
|
DBI::dbGetQuery(con, sprintf(
|
||||||
|
"SELECT COUNT(*) AS n FROM read_parquet(%s)",
|
||||||
|
.sql_quote(path)
|
||||||
|
))$n
|
||||||
|
}
|
||||||
|
|
||||||
|
#' @noRd
|
||||||
|
.sql_quote <- function(x) paste0("'", gsub("'", "''", x), "'")
|
||||||
|
|
||||||
|
# Hand-build manifest.json following the shape of the previous fixture
|
||||||
|
# manifest: schema_version / built_at / pipeline_commit / fixture_note /
|
||||||
|
# data_vintage / scope / schema / files.long_partitions / files.metadata /
|
||||||
|
# series_breaks_ref / reader_spec_ref. Every sha256 / row_count / size_bytes
|
||||||
|
# is freshly computed against the files actually written into fixture_dir.
|
||||||
|
#' @noRd
|
||||||
|
.build_fixture_manifest <- function(fixture_dir, source_manifest, years) {
|
||||||
|
long_partitions <- lapply(years, function(yr) {
|
||||||
|
rel <- file.path("data", "long", sprintf("year=%d", yr), "part-0.parquet")
|
||||||
|
path <- file.path(fixture_dir, rel)
|
||||||
|
list(
|
||||||
|
year = as.integer(yr),
|
||||||
|
path = gsub("\\\\", "/", rel),
|
||||||
|
sha256 = digest::digest(path, algo = "sha256", file = TRUE),
|
||||||
|
row_count = as.integer(.parquet_row_count(path)),
|
||||||
|
size_bytes = as.integer(file.info(path)$size)
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
metadata_files <- c(
|
||||||
|
"canonical_alias.parquet",
|
||||||
|
"canonical_fips_xwalk.parquet",
|
||||||
|
"summary_categories.parquet"
|
||||||
|
)
|
||||||
|
metadata <- lapply(metadata_files, function(f) {
|
||||||
|
rel <- file.path("data", f)
|
||||||
|
path <- file.path(fixture_dir, rel)
|
||||||
|
list(
|
||||||
|
path = gsub("\\\\", "/", rel),
|
||||||
|
sha256 = digest::digest(path, algo = "sha256", file = TRUE),
|
||||||
|
description = f
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
list(
|
||||||
|
schema_version = as.integer(source_manifest$schema_version),
|
||||||
|
built_at = format(Sys.time(), "%Y-%m-%dT%H:%M:%SZ", tz = "UTC"),
|
||||||
|
pipeline_commit = source_manifest$pipeline_commit,
|
||||||
|
fixture_note = paste(
|
||||||
|
"Two-year (2019-2020) fixture for uscogdata tests. Full corpus",
|
||||||
|
"available via USCOGDATA_URL. Regenerated for Phase P",
|
||||||
|
"(schema_version 4, uniformly 12-char canonical_govid) with the full",
|
||||||
|
"canonical_fips_xwalk master and the new canonical_alias lookup",
|
||||||
|
"table via data-raw/regenerate_fixture_corpus.R."
|
||||||
|
),
|
||||||
|
data_vintage = source_manifest$data_vintage,
|
||||||
|
scope = source_manifest$scope,
|
||||||
|
schema = source_manifest$schema,
|
||||||
|
files = list(
|
||||||
|
long_partitions = long_partitions,
|
||||||
|
metadata = metadata
|
||||||
|
),
|
||||||
|
series_breaks_ref = source_manifest$series_breaks_ref,
|
||||||
|
reader_spec_ref = source_manifest$reader_spec_ref
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
if (identical(environment(), globalenv()) && sys.nframe() == 0L) {
|
||||||
|
args <- commandArgs(trailingOnly = TRUE)
|
||||||
|
if (length(args) >= 1L) {
|
||||||
|
regenerate_fixture_corpus(publish_cache_dir = args[[1]])
|
||||||
|
} else {
|
||||||
|
regenerate_fixture_corpus()
|
||||||
|
}
|
||||||
|
}
|
||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+18
-46
@@ -1,54 +1,21 @@
|
|||||||
{
|
{
|
||||||
"schema_version": 3,
|
"schema_version": 4,
|
||||||
"built_at": "2026-04-29T18:57:18Z",
|
"built_at": "2026-07-11T13:24:01Z",
|
||||||
"pipeline_commit": "bd3e744",
|
"pipeline_commit": "1a00925",
|
||||||
"fixture_note": "Two-year (2019-2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated with Layer 1 (canonical_govid resolver gate) + Layer 2 (PID-era extended xwalk).",
|
"fixture_note": "Two-year (2019-2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated for Phase P (schema_version 4, uniformly 12-char canonical_govid) with the full canonical_fips_xwalk master and the new canonical_alias lookup table via data-raw/regenerate_fixture_corpus.R.",
|
||||||
"data_vintage": {
|
"data_vintage": {
|
||||||
"census_source_downloaded": "unknown",
|
"census_source_downloaded": "unknown",
|
||||||
"cpi_vintage": "FRED CPIAUCSL",
|
"cpi_vintage": "FRED CPIAUCSL",
|
||||||
"acs_vintage": "ACS 2018-2022 5-year"
|
"acs_vintage": "ACS 2018-2022 5-year"
|
||||||
},
|
},
|
||||||
"scope": {
|
"scope": {
|
||||||
"gov_types_included": [
|
"gov_types_included": [0, 1, 2, 3],
|
||||||
0,
|
"gov_types_excluded": [4, 5],
|
||||||
1,
|
|
||||||
2,
|
|
||||||
3
|
|
||||||
],
|
|
||||||
"gov_types_excluded": [
|
|
||||||
4,
|
|
||||||
5
|
|
||||||
],
|
|
||||||
"scope_note": "v0.1 covers state, county, city/municipality, and township governments. Special districts (type 4) and school districts (type 5) are excluded pending validation in a future cycle."
|
"scope_note": "v0.1 covers state, county, city/municipality, and township governments. Special districts (type 4) and school districts (type 5) are excluded pending validation in a future cycle."
|
||||||
},
|
},
|
||||||
"schema": {
|
"schema": {
|
||||||
"long_column_count": 24,
|
"long_column_count": 24,
|
||||||
"long_columns": [
|
"long_columns": ["fips_state", "type", "fips_county", "govid", "gov_blank", "gov_name", "county_name", "fips_state_code", "fips_county_code", "fips_place_code", "population", "popyear", "enrollment", "enrollyear", "function_code", "sch_level_code", "fiscal_year_end", "srvy_year", "item_code", "amt", "srv_data", "impute_flag", "is_aggregate", "canonical_govid"],
|
||||||
"fips_state",
|
|
||||||
"type",
|
|
||||||
"fips_county",
|
|
||||||
"govid",
|
|
||||||
"gov_blank",
|
|
||||||
"gov_name",
|
|
||||||
"county_name",
|
|
||||||
"fips_state_code",
|
|
||||||
"fips_county_code",
|
|
||||||
"fips_place_code",
|
|
||||||
"population",
|
|
||||||
"popyear",
|
|
||||||
"enrollment",
|
|
||||||
"enrollyear",
|
|
||||||
"function_code",
|
|
||||||
"sch_level_code",
|
|
||||||
"fiscal_year_end",
|
|
||||||
"srvy_year",
|
|
||||||
"item_code",
|
|
||||||
"amt",
|
|
||||||
"srv_data",
|
|
||||||
"impute_flag",
|
|
||||||
"is_aggregate",
|
|
||||||
"canonical_govid"
|
|
||||||
],
|
|
||||||
"data_dictionary": "docs/data_dictionary.md"
|
"data_dictionary": "docs/data_dictionary.md"
|
||||||
},
|
},
|
||||||
"files": {
|
"files": {
|
||||||
@@ -56,27 +23,32 @@
|
|||||||
{
|
{
|
||||||
"year": 2019,
|
"year": 2019,
|
||||||
"path": "data/long/year=2019/part-0.parquet",
|
"path": "data/long/year=2019/part-0.parquet",
|
||||||
"sha256": "d93affbbf9c46bd193fa5e07d89dfa9b08ad4f6fc2442fe7a27e6e4ef28f8bd9",
|
"sha256": "c0a2bf0758af129d5dfddb6ff6665cc435ddee87fd6879e788fb56ed53ab22b8",
|
||||||
"row_count": 318139,
|
"row_count": 318139,
|
||||||
"size_bytes": 1425125
|
"size_bytes": 1441404
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"year": 2020,
|
"year": 2020,
|
||||||
"path": "data/long/year=2020/part-0.parquet",
|
"path": "data/long/year=2020/part-0.parquet",
|
||||||
"sha256": "c9224833bd914cc85efb2ec62f83f75c34dfa469d4398f5cb1789fd028c03389",
|
"sha256": "92570b9d55ec3425d034db37838f91c3b8359d0454d3d98730a6016b62e4bb48",
|
||||||
"row_count": 317500,
|
"row_count": 317500,
|
||||||
"size_bytes": 1428049
|
"size_bytes": 1444011
|
||||||
}
|
}
|
||||||
],
|
],
|
||||||
"metadata": [
|
"metadata": [
|
||||||
|
{
|
||||||
|
"path": "data/canonical_alias.parquet",
|
||||||
|
"sha256": "db784d4ec9e5abb4b033405627d8bdd6f8d8ff96b3033a5b2587b4b59113ec9c",
|
||||||
|
"description": "canonical_alias.parquet"
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"path": "data/canonical_fips_xwalk.parquet",
|
"path": "data/canonical_fips_xwalk.parquet",
|
||||||
"sha256": "4bbdf0415b0ae1a24927869bb9e8e75eca5d7bcceebdf8c26346c7cbf8472d34",
|
"sha256": "c0b1295e779b601f2782de40ab143556324e74d30d214fc10997c2244c109389",
|
||||||
"description": "canonical_fips_xwalk.parquet"
|
"description": "canonical_fips_xwalk.parquet"
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"path": "data/summary_categories.parquet",
|
"path": "data/summary_categories.parquet",
|
||||||
"sha256": "60045e22bc2723318fa2cb73f8e5038250dc54d24b3447c6750dfe29035335b8",
|
"sha256": "dd59e7f58a022679ad43511c8c8e938b8dd4be81196bbeeee21e67bdcca2295b",
|
||||||
"description": "summary_categories.parquet"
|
"description": "summary_categories.parquet"
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
|
|||||||
Reference in New Issue
Block a user