From 8d2a8d341f9dc18e2e11547976fd7e0392ad56d9 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Thu, 23 Apr 2026 10:21:36 -0400 Subject: [PATCH] feat: view registration from inst/sql/ MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Seven DuckDB views register on session open: long (raw), spending_long and revenue_long (prefix-filtered, NOT is_aggregate per reader-spec §4), canonical_fips_xwalk and summary_categories (identity), spending_annotated and revenue_annotated (LEFT JOIN xwalk + categories for verb composition). File prefix `NN-` enforces creation order so *_annotated views resolve their *_long dependencies. Deviation from plan: series_breaks/cpi_annual/legacy_aggregate_map and *_with_transforms views are deferred — their backing parquets are not in the v0.1 corpus (manifest.files.metadata only lists canonical_fips_xwalk and summary_categories). The verbs will compute CPI adjustment verb-side against a bundled cpi table in a later task. --- inst/sql/.gitkeep | 0 inst/sql/10-long.sql | 3 ++ inst/sql/20-spending_long.sql | 5 +++ inst/sql/21-revenue_long.sql | 5 +++ inst/sql/30-canonical_fips_xwalk.sql | 3 ++ inst/sql/31-summary_categories.sql | 3 ++ inst/sql/40-spending_annotated.sql | 16 ++++++++ inst/sql/41-revenue_annotated.sql | 16 ++++++++ tests/testthat/test-views.R | 59 ++++++++++++++++++++++++++++ 9 files changed, 110 insertions(+) delete mode 100644 inst/sql/.gitkeep create mode 100644 inst/sql/10-long.sql create mode 100644 inst/sql/20-spending_long.sql create mode 100644 inst/sql/21-revenue_long.sql create mode 100644 inst/sql/30-canonical_fips_xwalk.sql create mode 100644 inst/sql/31-summary_categories.sql create mode 100644 inst/sql/40-spending_annotated.sql create mode 100644 inst/sql/41-revenue_annotated.sql create mode 100644 tests/testthat/test-views.R diff --git a/inst/sql/.gitkeep b/inst/sql/.gitkeep deleted file mode 100644 index e69de29..0000000 diff --git a/inst/sql/10-long.sql b/inst/sql/10-long.sql new file mode 100644 index 0000000..5cf5fcc --- /dev/null +++ b/inst/sql/10-long.sql @@ -0,0 +1,3 @@ +CREATE OR REPLACE VIEW long AS +SELECT * +FROM read_parquet('{url}data/long/**/*.parquet', hive_partitioning = true); diff --git a/inst/sql/20-spending_long.sql b/inst/sql/20-spending_long.sql new file mode 100644 index 0000000..71eeec2 --- /dev/null +++ b/inst/sql/20-spending_long.sql @@ -0,0 +1,5 @@ +CREATE OR REPLACE VIEW spending_long AS +SELECT * +FROM long +WHERE LEFT(item_code, 1) IN ('E', 'F', 'G', 'K') + AND NOT is_aggregate; diff --git a/inst/sql/21-revenue_long.sql b/inst/sql/21-revenue_long.sql new file mode 100644 index 0000000..013a701 --- /dev/null +++ b/inst/sql/21-revenue_long.sql @@ -0,0 +1,5 @@ +CREATE OR REPLACE VIEW revenue_long AS +SELECT * +FROM long +WHERE LEFT(item_code, 1) IN ('T', 'A', 'U', 'B', 'C', 'D') + AND NOT is_aggregate; diff --git a/inst/sql/30-canonical_fips_xwalk.sql b/inst/sql/30-canonical_fips_xwalk.sql new file mode 100644 index 0000000..77bf13d --- /dev/null +++ b/inst/sql/30-canonical_fips_xwalk.sql @@ -0,0 +1,3 @@ +CREATE OR REPLACE VIEW canonical_fips_xwalk AS +SELECT * +FROM read_parquet('{url}data/canonical_fips_xwalk.parquet'); diff --git a/inst/sql/31-summary_categories.sql b/inst/sql/31-summary_categories.sql new file mode 100644 index 0000000..0fa10df --- /dev/null +++ b/inst/sql/31-summary_categories.sql @@ -0,0 +1,3 @@ +CREATE OR REPLACE VIEW summary_categories AS +SELECT * +FROM read_parquet('{url}data/summary_categories.parquet'); diff --git a/inst/sql/40-spending_annotated.sql b/inst/sql/40-spending_annotated.sql new file mode 100644 index 0000000..e69b758 --- /dev/null +++ b/inst/sql/40-spending_annotated.sql @@ -0,0 +1,16 @@ +CREATE OR REPLACE VIEW spending_annotated AS +SELECT + s.*, + x.gov_name AS xwalk_gov_name, + x.govs_type, + x.type_label, + x.fips_state AS xwalk_fips_state, + x.fips_county AS xwalk_fips_county, + x.fips_place, + x.population_acs, + c.category, + c.category_type, + c.spend_subtype +FROM spending_long s +LEFT JOIN canonical_fips_xwalk x USING (canonical_govid) +LEFT JOIN summary_categories c USING (item_code); diff --git a/inst/sql/41-revenue_annotated.sql b/inst/sql/41-revenue_annotated.sql new file mode 100644 index 0000000..eec14bc --- /dev/null +++ b/inst/sql/41-revenue_annotated.sql @@ -0,0 +1,16 @@ +CREATE OR REPLACE VIEW revenue_annotated AS +SELECT + s.*, + x.gov_name AS xwalk_gov_name, + x.govs_type, + x.type_label, + x.fips_state AS xwalk_fips_state, + x.fips_county AS xwalk_fips_county, + x.fips_place, + x.population_acs, + c.category, + c.category_type, + c.revenue_subtype +FROM revenue_long s +LEFT JOIN canonical_fips_xwalk x USING (canonical_govid) +LEFT JOIN summary_categories c USING (item_code); diff --git a/tests/testthat/test-views.R b/tests/testthat/test-views.R new file mode 100644 index 0000000..dfb7a33 --- /dev/null +++ b/tests/testthat/test-views.R @@ -0,0 +1,59 @@ +test_that("all expected views register on session open", { + skip_if_no_corpus() + con <- cog_open() + on.exit(cog_close()) + views <- DBI::dbGetQuery(con, + "SELECT table_name FROM information_schema.tables + WHERE table_schema = 'main' AND table_type = 'VIEW'" + ) + expected <- c( + "long", "spending_long", "revenue_long", + "canonical_fips_xwalk", "summary_categories", + "spending_annotated", "revenue_annotated" + ) + expect_true(all(expected %in% views$table_name)) +}) + +test_that("spending_long filters to E/F/G/K prefixes and excludes aggregates", { + skip_if_no_corpus() + con <- cog_open() + on.exit(cog_close()) + prefixes <- DBI::dbGetQuery(con, + "SELECT DISTINCT LEFT(item_code, 1) AS pfx FROM spending_long" + )$pfx + expect_true(all(prefixes %in% c("E", "F", "G", "K"))) + + agg_count <- DBI::dbGetQuery(con, + "SELECT count(*) AS n FROM spending_long WHERE is_aggregate" + )$n + expect_equal(agg_count, 0) +}) + +test_that("revenue_long filters to T/A/U/B/C/D prefixes and excludes aggregates", { + skip_if_no_corpus() + con <- cog_open() + on.exit(cog_close()) + prefixes <- DBI::dbGetQuery(con, + "SELECT DISTINCT LEFT(item_code, 1) AS pfx FROM revenue_long" + )$pfx + expect_true(all(prefixes %in% c("T", "A", "U", "B", "C", "D"))) + + agg_count <- DBI::dbGetQuery(con, + "SELECT count(*) AS n FROM revenue_long WHERE is_aggregate" + )$n + expect_equal(agg_count, 0) +}) + +test_that("spending_annotated carries category + xwalk columns", { + skip_if_no_corpus() + con <- cog_open() + on.exit(cog_close()) + row <- DBI::dbGetQuery(con, + "SELECT * FROM spending_annotated LIMIT 1" + ) + for (nm in c("canonical_govid", "item_code", "amt", + "xwalk_gov_name", "govs_type", "population_acs", + "category", "spend_subtype")) { + expect_true(nm %in% names(row), info = paste("missing column:", nm)) + } +})