From d09bfd6aef3dcc8fd85651ceafcefc360992ecb7 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Mon, 3 Aug 2026 09:46:36 -0400 Subject: [PATCH] feat: register balance_long / balance_annotated behind a column gate (#25) --- R/views.R | 24 ++++++++++++++++ inst/sql/26-balance_long.sql | 22 +++++++++++++++ inst/sql/46-balance_annotated.sql | 16 +++++++++++ tests/testthat/helper-fixture.R | 33 ++++++++++++++++++++++ tests/testthat/test-balances.R | 47 +++++++++++++++++++++++++++++++ 5 files changed, 142 insertions(+) create mode 100644 inst/sql/26-balance_long.sql create mode 100644 inst/sql/46-balance_annotated.sql create mode 100644 tests/testthat/test-balances.R diff --git a/R/views.R b/R/views.R index 172d6e9..28bf42c 100644 --- a/R/views.R +++ b/R/views.R @@ -45,6 +45,29 @@ "37-code_set.sql" = "code_set.parquet" ) +# Cash and security holdings (uscogdata#25). 46- selects +# `c.balance_subtype`, a column that arrived with cog_pipeline #76/#77 and +# WITHOUT a schema_version bump -- so neither existing gate applies: +# .harmonization_view_files keys on schema_version, .representation_view_files +# on the presence of a FILE. Here the discriminator is a COLUMN on a table +# that exists either way. CREATE VIEW resolves its source schema eagerly, so +# on an older corpus 46- would fail at registration with "Binder Error: +# Referenced column balance_subtype not found" rather than at query time. +.balance_view_files <- c("26-balance_long.sql", "46-balance_annotated.sql") + +#' Does the mounted corpus's `summary_categories` carry `balance_subtype`? +#' Probed against the live connection rather than the manifest, because the +#' manifest describes files, not columns. +#' @noRd +.corpus_has_balance_subtype <- function(con) { + n <- DBI::dbGetQuery(con, + "SELECT COUNT(*) AS n FROM information_schema.columns + WHERE table_name = 'summary_categories' + AND column_name = 'balance_subtype'" + )$n + isTRUE(as.integer(n) > 0L) +} + #' Does the mounted corpus publish `file` (e.g. "code_set.parquet")? #' Reads the manifest's metadata list rather than stat-ing the URL, so it #' works identically for a local fixture and a remote share. @@ -66,6 +89,7 @@ if (base %in% .harmonization_view_files && schema_version < 5L) next if (base %in% names(.representation_view_files) && !.corpus_has_table(manifest, .representation_view_files[[base]])) next + if (base %in% .balance_view_files && !.corpus_has_balance_subtype(con)) next sql <- paste(readLines(f, warn = FALSE), collapse = "\n") sql <- gsub("\\{url\\}", url, sql, fixed = FALSE) DBI::dbExecute(con, sql) diff --git a/inst/sql/26-balance_long.sql b/inst/sql/26-balance_long.sql new file mode 100644 index 0000000..22095e8 --- /dev/null +++ b/inst/sql/26-balance_long.sql @@ -0,0 +1,22 @@ +-- Cash and security holdings, classified by crosswalk MEMBERSHIP on +-- category_type (see 21-revenue_long.sql for why first-letter prefixes cannot +-- do this job -- the X and Y families each span revenue, expenditure AND +-- balance). +-- +-- These rows are STOCKS: a balance at a point in time, not a flow over a +-- fiscal year. Summing a stock with a flow is meaningless, which is why they +-- live behind a third view rather than as a subtype of either money view, and +-- why neither spending_long nor revenue_long can reach them. +-- +-- `NOT is_aggregate` mirrors spending_long / revenue_long. The wide-era +-- aggregate-only holdings codes (X40/X41) are deliberately outside this view; +-- they are reachable only through the recipe path, which bypasses this filter +-- by design (cog_pipeline/docs/phase_r_harmonization_review.md ยง 0.2). +CREATE OR REPLACE VIEW balance_long AS +SELECT * +FROM long +WHERE item_code IN ( + SELECT item_code FROM summary_categories + WHERE category_type = 'balance' + ) + AND NOT is_aggregate; diff --git a/inst/sql/46-balance_annotated.sql b/inst/sql/46-balance_annotated.sql new file mode 100644 index 0000000..8a40a91 --- /dev/null +++ b/inst/sql/46-balance_annotated.sql @@ -0,0 +1,16 @@ +CREATE OR REPLACE VIEW balance_annotated AS +SELECT + s.*, + x.gov_name AS xwalk_gov_name, + x.govs_type, + x.type_label, + x.fips_state AS xwalk_fips_state, + x.fips_county AS xwalk_fips_county, + x.fips_place, + x.population_acs, + c.category, + c.category_type, + c.balance_subtype +FROM balance_long s +LEFT JOIN canonical_fips_xwalk x USING (canonical_govid) +LEFT JOIN summary_categories c USING (item_code); diff --git a/tests/testthat/helper-fixture.R b/tests/testthat/helper-fixture.R index 3ea3744..9d23512 100644 --- a/tests/testthat/helper-fixture.R +++ b/tests/testthat/helper-fixture.R @@ -154,3 +154,36 @@ with_corpus_missing_ig_categories <- function(code) { }, add = TRUE) force(code) } + +# Copy the bundled fixture to a temp dir with summary_categories.parquet +# rewritten to DROP the balance_subtype column, then run `code` against it. +# Models a corpus published before cog_pipeline #76/#77. schema_version is +# left untouched deliberately: that change shipped without a version bump, so +# column presence is the only honest signal -- this helper is what proves the +# package keys off it. Mirrors with_corpus_missing_ig_categories(). +with_corpus_missing_balance_subtype <- function(code) { + src <- fixture_corpus_path() + tmp <- withr::local_tempdir(.local_envir = parent.frame()) + file.copy(list.files(src, full.names = TRUE), tmp, recursive = TRUE) + + cats_path <- file.path(tmp, "data", "summary_categories.parquet") + filtered_path <- file.path(tmp, "data", "summary_categories_filtered.parquet") + write_con <- DBI::dbConnect(duckdb::duckdb()) + on.exit(DBI::dbDisconnect(write_con, shutdown = TRUE), add = TRUE) + DBI::dbExecute(write_con, sprintf( + "COPY (SELECT * EXCLUDE (balance_subtype) FROM read_parquet(%s)) + TO %s (FORMAT PARQUET)", + uscogdata:::.sql_lit_chr(cats_path), uscogdata:::.sql_lit_chr(filtered_path) + )) + file.remove(cats_path) + file.rename(filtered_path, cats_path) + + old_url <- Sys.getenv("USCOGDATA_URL", unset = NA) + uscogdata:::cog_close() + Sys.setenv(USCOGDATA_URL = paste0(tmp, "/")) + on.exit({ + uscogdata:::cog_close() + if (is.na(old_url)) Sys.unsetenv("USCOGDATA_URL") else Sys.setenv(USCOGDATA_URL = old_url) + }, add = TRUE) + force(code) +} diff --git a/tests/testthat/test-balances.R b/tests/testthat/test-balances.R new file mode 100644 index 0000000..01fe6ab --- /dev/null +++ b/tests/testthat/test-balances.R @@ -0,0 +1,47 @@ +test_that("balance views register and carry only balance codes", { + skip_if_no_corpus() + con <- cog_open() + on.exit(cog_close()) + + views <- DBI::dbGetQuery(con, + "SELECT table_name FROM information_schema.tables + WHERE table_schema = 'main' AND table_type = 'VIEW'" + )$table_name + expect_true(all(c("balance_long", "balance_annotated") %in% views)) + + # Every item_code in balance_long is a category_type = 'balance' member. + leak <- DBI::dbGetQuery(con, + "SELECT COUNT(*) AS n FROM balance_long + WHERE item_code NOT IN ( + SELECT item_code FROM summary_categories WHERE category_type = 'balance')" + )$n + expect_identical(as.integer(leak), 0L) + + # And no aggregate row survives, mirroring revenue_long. + agg <- DBI::dbGetQuery(con, + "SELECT COUNT(*) AS n FROM balance_long WHERE is_aggregate" + )$n + expect_identical(as.integer(agg), 0L) + + # balance_annotated exposes the subtype column the verb groups on. + cols <- DBI::dbGetQuery(con, + "SELECT column_name FROM information_schema.columns + WHERE table_name = 'balance_annotated'" + )$column_name + expect_true(all(c("category", "category_type", "balance_subtype") %in% cols)) +}) + +test_that("balance views are skipped on a corpus without balance_subtype", { + skip_if_no_corpus() + with_corpus_missing_balance_subtype({ + con <- cog_open() + on.exit(cog_close()) + views <- DBI::dbGetQuery(con, + "SELECT table_name FROM information_schema.tables + WHERE table_schema = 'main' AND table_type = 'VIEW'" + )$table_name + # Registration must SKIP them, not error -- an older corpus stays usable. + expect_false(any(c("balance_long", "balance_annotated") %in% views)) + expect_true("revenue_long" %in% views) + }) +})