From c682e6547d7ca2ba4325f9f66fb45eff34615643 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Fri, 24 Apr 2026 09:49:09 -0400 Subject: [PATCH] feat: cog_spending + cog_revenue + cog_explain Three core query verbs over the spending_annotated / revenue_annotated DuckDB views. Each verb accepts vector govid, vector years, optional category filter, per_capita flag, and adjust_to_year for CPI-U real-dollar conversion (bundled index). Amounts are returned in full USD (SUM(amt) * 1000) so callers can freely rescale to millions/billions. The $1,000s -> $USD conversion is recorded in provenance$transformations$units_conversion. Every result carries an attr(., "provenance") list matching inst/schemas/provenance-v1.json. cog_explain() prints the structured form via cli or returns the raw list for MCP/JSON consumers. Also: .fetch_or_cache_manifest() now handles local fixture paths so tests can point USCOGDATA_FIXTURE_URL at the pipeline publish_cache/ without a working HTTP server. Tests: 80 pass / 0 fail. devtools::check() 0E/0W/2N (both notes pre-existing / environmental). --- NAMESPACE | 3 + R/explain.R | 100 ++++++++++++++++++ R/manifest.R | 18 +++- R/provenance.R | 83 +++++++++++++++ R/revenue.R | 29 ++++++ R/spending.R | 181 +++++++++++++++++++++++++++++++++ man/cog_explain.Rd | 23 +++++ man/cog_revenue.Rd | 41 ++++++++ man/cog_spending.Rd | 43 ++++++++ tests/testthat/test-explain.R | 32 ++++++ tests/testthat/test-revenue.R | 38 +++++++ tests/testthat/test-spending.R | 82 +++++++++++++++ 12 files changed, 672 insertions(+), 1 deletion(-) create mode 100644 R/explain.R create mode 100644 R/provenance.R create mode 100644 R/revenue.R create mode 100644 R/spending.R create mode 100644 man/cog_explain.Rd create mode 100644 man/cog_revenue.Rd create mode 100644 man/cog_spending.Rd create mode 100644 tests/testthat/test-explain.R create mode 100644 tests/testthat/test-revenue.R create mode 100644 tests/testthat/test-spending.R diff --git a/NAMESPACE b/NAMESPACE index 6ae9268..ac2f7cf 100644 --- a/NAMESPACE +++ b/NAMESPACE @@ -1,2 +1,5 @@ # Generated by roxygen2: do not edit by hand +export(cog_explain) +export(cog_revenue) +export(cog_spending) diff --git a/R/explain.R b/R/explain.R new file mode 100644 index 0000000..7bf3961 --- /dev/null +++ b/R/explain.R @@ -0,0 +1,100 @@ +# R/explain.R + +#' Explain a verb result's provenance +#' +#' Prints the structured provenance attached to a tibble returned by any +#' `cog_*` verb, or returns it as a list for downstream use (MCP tools, +#' dashboards, JSON export). +#' +#' @param result A tibble returned by a `cog_*` verb. +#' @param format `"print"` (default) for a human-readable cli summary; +#' returns `result` invisibly for chaining. `"list"` returns the raw +#' provenance list (identical to `attr(result, "provenance")`). +#' @return Either `result` (invisibly) or the provenance list. +#' @export +cog_explain <- function(result, format = c("print", "list")) { + format <- match.arg(format) + prov <- attr(result, "provenance") + if (is.null(prov)) { + cli::cli_abort(c( + "No `provenance` attribute on result.", + i = "Pass a tibble returned by a cog_* verb (e.g. cog_spending())." + )) + } + if (format == "list") return(prov) + .print_provenance(prov) + invisible(result) +} + +#' @noRd +.print_provenance <- function(prov) { + cli::cli_h1("{prov$verb}()") + + tgt_ids <- paste(prov$target$canonical_govid, collapse = ", ") + tgt_names <- if (length(prov$target$gov_name) == 0L) { + "(no rows returned)" + } else { + paste(prov$target$gov_name, collapse = ", ") + } + cli::cli_text("Target: {tgt_names} [canonical_govid: {tgt_ids}]") + + yrs <- prov$years + cli::cli_text(if (length(yrs) == 1L) { + "Year: {yrs}" + } else { + "Years: {min(yrs)}-{max(yrs)} ({length(yrs)} years)" + }) + + if (!is.null(prov$category)) { + cli::cli_text("Category: {paste(prov$category, collapse = ', ')}") + } else { + cli::cli_text("Category: (all)") + } + + cli::cli_h2("Codes observed") + codes <- prov$codes_summed$observed + if (length(codes) == 0L) { + cli::cli_alert_info("No item codes matched.") + } else { + cli::cli_ul(codes) + } + + if (isTRUE(prov$aggregate_fallback$applied)) { + cli::cli_h2("Aggregate fallback") + cli::cli_alert_warning( + "Aggregate fallback used for years: {paste(prov$aggregate_fallback$years, collapse = ', ')}" + ) + } + + cli::cli_h2("Transformations") + uc <- prov$transformations$units_conversion + if (isTRUE(uc$applied)) { + cli::cli_text("Units: {uc$source_unit} -> {uc$target_unit} (x{uc$multiplier})") + } + pc <- prov$transformations$per_capita + if (isTRUE(pc$applied)) { + cli::cli_text("Per-capita denominator: {pc$denominator_source}") + } + infl <- prov$transformations$inflation + if (isTRUE(infl$applied)) { + cli::cli_text("Inflation: {infl$index}, base year {infl$base_year}") + } + + cli::cli_h2("Scope") + cli::cli_text( + "Included gov types: {paste(prov$scope$gov_types_included, collapse = ', ')}" + ) + cli::cli_text( + "Excluded gov types: {paste(prov$scope$gov_types_excluded, collapse = ', ')}" + ) + if (nzchar(prov$scope$scope_note %||% "")) { + cli::cli_text("Note: {prov$scope$scope_note}") + } + + cli::cli_h2("Data vintage") + cli::cli_text( + "Manifest schema v{prov$manifest$schema_version}, pipeline {prov$manifest$pipeline_commit}, built {prov$manifest$built_at}" + ) + + invisible(NULL) +} diff --git a/R/manifest.R b/R/manifest.R index 9fa7c0d..bd2caf1 100644 --- a/R/manifest.R +++ b/R/manifest.R @@ -1,8 +1,19 @@ # R/manifest.R -#' Fetch manifest.json from URL, cache locally, validate TTL. +#' Fetch manifest.json from URL (or read from a local fixture path), +#' cache locally, validate TTL. #' @noRd .fetch_or_cache_manifest <- function(url, cache_dir) { + # Local fixture path: read manifest directly; skip cache/TTL plumbing so + # tests pick up regenerated manifests immediately. + if (.is_local_path(url)) { + local_manifest <- file.path(url, "manifest.json") + if (!file.exists(local_manifest)) { + cli::cli_abort("Local fixture has no manifest.json at {local_manifest}") + } + return(jsonlite::fromJSON(local_manifest, simplifyVector = FALSE)) + } + cache_path <- file.path(cache_dir, "manifest.json") ttl <- as.integer(.cfg("manifest_ttl_secs")) @@ -19,6 +30,11 @@ jsonlite::fromJSON(cache_path, simplifyVector = FALSE) } +#' @noRd +.is_local_path <- function(url) { + !grepl("^[a-zA-Z][a-zA-Z0-9+.-]*://", url) +} + #' @noRd .validate_schema <- function(manifest, expected_version) { if (manifest$schema_version != expected_version) { diff --git a/R/provenance.R b/R/provenance.R new file mode 100644 index 0000000..590d56d --- /dev/null +++ b/R/provenance.R @@ -0,0 +1,83 @@ +# R/provenance.R +# Shared provenance construction. Matches inst/schemas/provenance-v1.json. + +#' @noRd +.build_provenance <- function(verb, call, govid, years, category, + per_capita, adjust_to_year, result, sql, + subtype_col) { + manifest <- .uscogdata_env$manifest + + codes <- result[["codes_included"]] + codes_observed <- if (length(codes) == 0L) { + character(0) + } else { + sorted <- sort(unique(unlist(strsplit(codes, ",", fixed = TRUE)))) + sorted[nzchar(sorted)] + } + + agg_flag <- result[["aggregate_fallback"]] + agg_applied <- isTRUE(any(agg_flag, na.rm = TRUE)) + agg_years <- if (agg_applied) { + unique(as.integer(result$year[which(agg_flag)])) + } else { + integer(0) + } + + gov_names <- if (nrow(result) == 0L) { + character(0) + } else { + unique(result$gov_name) + } + + list( + verb = verb, + call = paste(deparse(call), collapse = " "), + target = list( + canonical_govid = as.character(govid), + gov_name = gov_names + ), + years = as.integer(years), + category = category, + scope = list( + gov_types_included = as.integer(unlist(manifest$scope$gov_types_included)), + gov_types_excluded = as.integer(unlist(manifest$scope$gov_types_excluded)), + scope_note = manifest$scope$scope_note %||% "" + ), + codes_summed = list( + observed = codes_observed, + subtype_column = subtype_col + ), + aggregate_fallback = list( + applied = agg_applied, + years = agg_years + ), + transformations = list( + units_conversion = list( + applied = TRUE, + source_unit = "$1,000s (raw Census)", + target_unit = "$USD", + multiplier = 1000L + ), + per_capita = list( + applied = isTRUE(per_capita), + denominator_source = if (isTRUE(per_capita)) { + "ACS 2018-2022 B01003_001 (population_acs from canonical_fips_xwalk)" + } else { + NA_character_ + } + ), + inflation = list( + applied = !is.null(adjust_to_year), + base_year = if (is.null(adjust_to_year)) NA_integer_ else as.integer(adjust_to_year), + index = if (is.null(adjust_to_year)) NA_character_ else "CPI-U (BLS CPIAUCSL annual average, bundled)" + ) + ), + series_break_refs = character(0), + manifest = list( + schema_version = as.integer(manifest$schema_version), + pipeline_commit = manifest$pipeline_commit %||% NA_character_, + built_at = manifest$built_at %||% NA_character_ + ), + sql_query = sql + ) +} diff --git a/R/revenue.R b/R/revenue.R new file mode 100644 index 0000000..393cf3c --- /dev/null +++ b/R/revenue.R @@ -0,0 +1,29 @@ +# R/revenue.R + +#' Summarized revenue by category +#' +#' Mirror of [cog_spending()] for revenue categories. One row per +#' `(year, canonical_govid, revenue_subtype, category)`. Amounts are returned +#' in **full U.S. dollars** (raw Census values are in $1,000s; this verb +#' multiplies by 1000 and records the conversion in `provenance`). +#' +#' @inheritParams cog_spending +#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`, +#' `revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`, +#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`, +#' `codes_included`, `aggregate_fallback`, `notes`. +#' @export +cog_revenue <- function(govid, years, category = NULL, + per_capita = FALSE, adjust_to_year = NULL) { + .verb_spendrev( + verb = "cog_revenue", + view = "revenue_annotated", + subtype_col = "revenue_subtype", + call = match.call(), + govid = govid, + years = years, + category = category, + per_capita = per_capita, + adjust_to_year = adjust_to_year + ) +} diff --git a/R/spending.R b/R/spending.R new file mode 100644 index 0000000..4cb5dc4 --- /dev/null +++ b/R/spending.R @@ -0,0 +1,181 @@ +# R/spending.R + +#' Summarized spending by category +#' +#' One row per `(year, canonical_govid, spend_subtype, category)`. Amounts are +#' returned in **full U.S. dollars** (the raw corpus stores them in $1,000s; +#' this verb multiplies by 1000 so downstream code can freely rescale to +#' millions/billions). The conversion is recorded in the provenance attribute +#' under `transformations$units_conversion`. +#' +#' @param govid Character vector of `canonical_govid` values. +#' @param years Integer vector of years. +#' @param category Character vector of category names (from +#' `summary_categories.category`), or `NULL` for all categories. +#' @param per_capita If `TRUE`, adds `amt_per_capita_nominal` (and +#' `amt_per_capita_real` when `adjust_to_year` is set) using +#' `population_acs` from the canonical xwalk. +#' @param adjust_to_year Integer base year for CPI-U real-dollar conversion, +#' or `NULL` for nominal only. +#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`, +#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real`, +#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`, +#' `codes_included`, `aggregate_fallback`, `notes`. Carries a `provenance` +#' attribute matching `inst/schemas/provenance-v1.json`. +#' @export +cog_spending <- function(govid, years, category = NULL, + per_capita = FALSE, adjust_to_year = NULL) { + .verb_spendrev( + verb = "cog_spending", + view = "spending_annotated", + subtype_col = "spend_subtype", + call = match.call(), + govid = govid, + years = years, + category = category, + per_capita = per_capita, + adjust_to_year = adjust_to_year + ) +} + +#' @noRd +.verb_spendrev <- function(verb, view, subtype_col, call, + govid, years, category, + per_capita, adjust_to_year) { + .validate_verb_inputs(govid, years, category, per_capita, adjust_to_year) + + govid <- as.character(govid) + years <- as.integer(years) + if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year) + + con <- .ensure_session() + + sql <- .build_verb_sql(view, subtype_col, govid, years, category) + result <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) + + if (per_capita) result <- .attach_per_capita(result, con, govid) + if (!is.null(adjust_to_year)) { + result <- .attach_real_dollars(result, adjust_to_year, per_capita) + } + + result$notes <- .notes_column(result) + + attr(result, "provenance") <- .build_provenance( + verb = verb, + call = call, + govid = govid, + years = years, + category = category, + per_capita = per_capita, + adjust_to_year = adjust_to_year, + result = result, + sql = sql, + subtype_col = subtype_col + ) + result +} + +#' @noRd +.validate_verb_inputs <- function(govid, years, category, + per_capita, adjust_to_year) { + if (!is.character(govid) || length(govid) == 0L) { + cli::cli_abort("`govid` must be a non-empty character vector.") + } + if (!(is.integer(years) || is.numeric(years)) || length(years) == 0L) { + cli::cli_abort("`years` must be a non-empty integer vector.") + } + if (!is.null(category) && !is.character(category)) { + cli::cli_abort("`category` must be character or NULL.") + } + if (!is.logical(per_capita) || length(per_capita) != 1L) { + cli::cli_abort("`per_capita` must be a length-1 logical.") + } + if (!is.null(adjust_to_year)) { + if (!(is.integer(adjust_to_year) || is.numeric(adjust_to_year)) || + length(adjust_to_year) != 1L) { + cli::cli_abort("`adjust_to_year` must be NULL or a length-1 integer.") + } + } + invisible(TRUE) +} + +#' @noRd +.sql_lit_chr <- function(x) { + safe <- gsub("'", "''", x, fixed = TRUE) + paste0("'", safe, "'", collapse = ",") +} + +#' @noRd +.build_verb_sql <- function(view, subtype_col, govid, years, category) { + govid_lit <- .sql_lit_chr(govid) + years_lit <- paste(as.integer(years), collapse = ",") + category_pred <- if (is.null(category)) { + "" + } else { + sprintf("AND category IN (%s)", .sql_lit_chr(category)) + } + + sprintf( + "SELECT + year, + canonical_govid, + COALESCE(xwalk_gov_name, gov_name) AS gov_name, + %1$s, + category, + SUM(amt) * 1000.0 AS amt_nominal, + string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS codes_included, + bool_and(is_aggregate) AS aggregate_fallback + FROM %2$s + WHERE canonical_govid IN (%3$s) + AND year IN (%4$s) + %5$s + GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, %1$s, category + ORDER BY year, canonical_govid, %1$s, category", + subtype_col, view, govid_lit, years_lit, category_pred + ) +} + +#' @noRd +.attach_per_capita <- function(result, con, govid) { + if (nrow(result) == 0L) { + result$amt_per_capita_nominal <- numeric(0) + return(result) + } + sql <- sprintf( + "SELECT canonical_govid, population_acs + FROM canonical_fips_xwalk + WHERE canonical_govid IN (%s)", + .sql_lit_chr(govid) + ) + pops <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) + result <- dplyr::left_join(result, pops, by = "canonical_govid") + result$amt_per_capita_nominal <- result$amt_nominal / result$population_acs + result$population_acs <- NULL + result +} + +#' @noRd +.attach_real_dollars <- function(result, adjust_to_year, per_capita) { + if (nrow(result) == 0L) { + result$amt_real <- numeric(0) + if (per_capita) result$amt_per_capita_real <- numeric(0) + return(result) + } + result$amt_real <- .inflate(result$amt_nominal, result$year, adjust_to_year) + if (per_capita && "amt_per_capita_nominal" %in% names(result)) { + result$amt_per_capita_real <- .inflate( + result$amt_per_capita_nominal, result$year, adjust_to_year + ) + } + result +} + +#' @noRd +.notes_column <- function(result) { + if (nrow(result) == 0L) return(character(0)) + ifelse( + isTRUE(result$aggregate_fallback) | result$aggregate_fallback %in% TRUE, + "Aggregate fallback applied; see cog_explain()", + "" + ) +} diff --git a/man/cog_explain.Rd b/man/cog_explain.Rd new file mode 100644 index 0000000..d67eb71 --- /dev/null +++ b/man/cog_explain.Rd @@ -0,0 +1,23 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/explain.R +\name{cog_explain} +\alias{cog_explain} +\title{Explain a verb result's provenance} +\usage{ +cog_explain(result, format = c("print", "list")) +} +\arguments{ +\item{result}{A tibble returned by a `cog_*` verb.} + +\item{format}{`"print"` (default) for a human-readable cli summary; +returns `result` invisibly for chaining. `"list"` returns the raw +provenance list (identical to `attr(result, "provenance")`).} +} +\value{ +Either `result` (invisibly) or the provenance list. +} +\description{ +Prints the structured provenance attached to a tibble returned by any +`cog_*` verb, or returns it as a list for downstream use (MCP tools, +dashboards, JSON export). +} diff --git a/man/cog_revenue.Rd b/man/cog_revenue.Rd new file mode 100644 index 0000000..96f1f3f --- /dev/null +++ b/man/cog_revenue.Rd @@ -0,0 +1,41 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/revenue.R +\name{cog_revenue} +\alias{cog_revenue} +\title{Summarized revenue by category} +\usage{ +cog_revenue( + govid, + years, + category = NULL, + per_capita = FALSE, + adjust_to_year = NULL +) +} +\arguments{ +\item{govid}{Character vector of `canonical_govid` values.} + +\item{years}{Integer vector of years.} + +\item{category}{Character vector of category names (from +`summary_categories.category`), or `NULL` for all categories.} + +\item{per_capita}{If `TRUE`, adds `amt_per_capita_nominal` (and +`amt_per_capita_real` when `adjust_to_year` is set) using +`population_acs` from the canonical xwalk.} + +\item{adjust_to_year}{Integer base year for CPI-U real-dollar conversion, +or `NULL` for nominal only.} +} +\value{ +Tibble with columns `year`, `canonical_govid`, `gov_name`, + `revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`, + optional `amt_per_capita_nominal`, optional `amt_per_capita_real`, + `codes_included`, `aggregate_fallback`, `notes`. +} +\description{ +Mirror of [cog_spending()] for revenue categories. One row per +`(year, canonical_govid, revenue_subtype, category)`. Amounts are returned +in **full U.S. dollars** (raw Census values are in $1,000s; this verb +multiplies by 1000 and records the conversion in `provenance`). +} diff --git a/man/cog_spending.Rd b/man/cog_spending.Rd new file mode 100644 index 0000000..36f60fc --- /dev/null +++ b/man/cog_spending.Rd @@ -0,0 +1,43 @@ +% Generated by roxygen2: do not edit by hand +% Please edit documentation in R/spending.R +\name{cog_spending} +\alias{cog_spending} +\title{Summarized spending by category} +\usage{ +cog_spending( + govid, + years, + category = NULL, + per_capita = FALSE, + adjust_to_year = NULL +) +} +\arguments{ +\item{govid}{Character vector of `canonical_govid` values.} + +\item{years}{Integer vector of years.} + +\item{category}{Character vector of category names (from +`summary_categories.category`), or `NULL` for all categories.} + +\item{per_capita}{If `TRUE`, adds `amt_per_capita_nominal` (and +`amt_per_capita_real` when `adjust_to_year` is set) using +`population_acs` from the canonical xwalk.} + +\item{adjust_to_year}{Integer base year for CPI-U real-dollar conversion, +or `NULL` for nominal only.} +} +\value{ +Tibble with columns `year`, `canonical_govid`, `gov_name`, + `spend_subtype`, `category`, `amt_nominal`, optional `amt_real`, + optional `amt_per_capita_nominal`, optional `amt_per_capita_real`, + `codes_included`, `aggregate_fallback`, `notes`. Carries a `provenance` + attribute matching `inst/schemas/provenance-v1.json`. +} +\description{ +One row per `(year, canonical_govid, spend_subtype, category)`. Amounts are +returned in **full U.S. dollars** (the raw corpus stores them in $1,000s; +this verb multiplies by 1000 so downstream code can freely rescale to +millions/billions). The conversion is recorded in the provenance attribute +under `transformations$units_conversion`. +} diff --git a/tests/testthat/test-explain.R b/tests/testthat/test-explain.R new file mode 100644 index 0000000..ee4b2c1 --- /dev/null +++ b/tests/testthat/test-explain.R @@ -0,0 +1,32 @@ +test_that("cog_explain prints verb header and target", { + skip_if_no_corpus() + r <- cog_spending("101006006", 2020L, "Corrections") + # cli writes to stderr; capture both stdout and message streams. + txt <- paste(c( + capture.output(cog_explain(r)), + capture.output(cog_explain(r), type = "message") + ), collapse = "\n") + expect_true(grepl("cog_spending", txt)) + expect_true(grepl("Corrections", txt)) + expect_true(grepl("101006006", txt)) +}) + +test_that("cog_explain format='list' returns structured provenance", { + skip_if_no_corpus() + r <- cog_spending("101006006", 2020L, "Corrections") + prov <- cog_explain(r, format = "list") + expect_identical(prov, attr(r, "provenance")) +}) + +test_that("cog_explain returns result invisibly for chaining", { + skip_if_no_corpus() + r <- cog_spending("101006006", 2020L, "Corrections") + res <- withVisible(cog_explain(r)) + expect_false(res$visible) + expect_identical(res$value, r) +}) + +test_that("cog_explain errors on non-verb input", { + df <- tibble::tibble(a = 1) + expect_error(cog_explain(df), "provenance") +}) diff --git a/tests/testthat/test-revenue.R b/tests/testthat/test-revenue.R new file mode 100644 index 0000000..7c0cdaf --- /dev/null +++ b/tests/testthat/test-revenue.R @@ -0,0 +1,38 @@ +test_that("cog_revenue returns expected shape for Broward Property Tax 2020", { + skip_if_no_corpus() + r <- cog_revenue("101006006", years = 2020L, category = "Property Tax") + expect_s3_class(r, "tbl_df") + expected_cols <- c("year", "canonical_govid", "gov_name", "revenue_subtype", + "category", "amt_nominal", "codes_included", + "aggregate_fallback", "notes") + expect_true(all(expected_cols %in% names(r))) + expect_equal(unique(r$canonical_govid), "101006006") + expect_equal(unique(r$year), 2020L) +}) + +test_that("cog_revenue with no category filter returns multiple categories", { + skip_if_no_corpus() + r <- cog_revenue("101006006", years = 2020L) + expect_gt(length(unique(r$category)), 1L) +}) + +test_that("cog_revenue with per_capita + adjust_to_year adds all columns", { + skip_if_no_corpus() + r <- cog_revenue("101006006", 2020L, + per_capita = TRUE, adjust_to_year = 2022L) + expect_true(all(c("amt_nominal", "amt_real", + "amt_per_capita_nominal", "amt_per_capita_real") %in% + names(r))) +}) + +test_that("cog_revenue result has provenance attribute", { + skip_if_no_corpus() + r <- cog_revenue("101006006", 2020L) + prov <- attr(r, "provenance") + expect_equal(prov$verb, "cog_revenue") + expect_true(grepl("revenue_annotated", prov$sql_query)) +}) + +test_that("cog_revenue rejects invalid inputs", { + expect_error(cog_revenue(123, 2020L), "character") +}) diff --git a/tests/testthat/test-spending.R b/tests/testthat/test-spending.R new file mode 100644 index 0000000..80c72dd --- /dev/null +++ b/tests/testthat/test-spending.R @@ -0,0 +1,82 @@ +test_that("cog_spending returns expected shape for Broward Corrections 2020", { + skip_if_no_corpus() + r <- cog_spending("101006006", years = 2020L, category = "Corrections") + expect_s3_class(r, "tbl_df") + expected_cols <- c("year", "canonical_govid", "gov_name", "spend_subtype", + "category", "amt_nominal", "codes_included", + "aggregate_fallback", "notes") + expect_true(all(expected_cols %in% names(r))) + expect_equal(unique(r$canonical_govid), "101006006") + expect_equal(unique(r$year), 2020L) + expect_equal(unique(r$category), "Corrections") + expect_true(all(r$spend_subtype %in% c("operations", "capital"))) + expect_true(all(r$amt_nominal > 0)) +}) + +test_that("cog_spending vectorised years + categories", { + skip_if_no_corpus() + r <- cog_spending("101006006", 2019:2020, + category = c("Corrections", "Police")) + expect_true(all(r$year %in% 2019:2020)) + expect_true(all(r$category %in% c("Corrections", "Police"))) + expect_gte(nrow(r), 4L) +}) + +test_that("cog_spending with per_capita adds per-capita nominal column", { + skip_if_no_corpus() + r <- cog_spending("101006006", 2020L, "Corrections", per_capita = TRUE) + expect_true("amt_per_capita_nominal" %in% names(r)) + expect_false("amt_real" %in% names(r)) + expect_false("amt_per_capita_real" %in% names(r)) + expect_true(all(is.finite(r$amt_per_capita_nominal))) + expect_true(all(r$amt_per_capita_nominal < r$amt_nominal)) +}) + +test_that("cog_spending with adjust_to_year adds real column", { + skip_if_no_corpus() + r <- cog_spending("101006006", 2015:2020, "Corrections", + adjust_to_year = 2022L) + expect_true("amt_real" %in% names(r)) + r2015 <- dplyr::filter(r, year == 2015L) + expect_true(any(r2015$amt_nominal != r2015$amt_real)) +}) + +test_that("cog_spending with per_capita + adjust_to_year adds all columns", { + skip_if_no_corpus() + r <- cog_spending("101006006", 2020L, "Corrections", + per_capita = TRUE, adjust_to_year = 2022L) + expect_true(all(c("amt_nominal", "amt_real", + "amt_per_capita_nominal", "amt_per_capita_real") %in% + names(r))) +}) + +test_that("cog_spending for unknown govid returns empty tibble", { + skip_if_no_corpus() + r <- cog_spending("XXXINVALID", 2020L, "Corrections") + expect_s3_class(r, "tbl_df") + expect_equal(nrow(r), 0L) + expect_true("notes" %in% names(r)) + # provenance still attached + expect_false(is.null(attr(r, "provenance"))) +}) + +test_that("cog_spending result has provenance attribute matching schema", { + skip_if_no_corpus() + r <- cog_spending("101006006", 2020L, "Corrections") + prov <- attr(r, "provenance") + expect_type(prov, "list") + expect_equal(prov$verb, "cog_spending") + required <- c("verb", "target", "years", "scope", "manifest", "sql_query") + expect_true(all(required %in% names(prov))) + expect_equal(prov$years, 2020L) + expect_equal(prov$category, "Corrections") + expect_type(prov$sql_query, "character") + expect_true(grepl("spending_annotated", prov$sql_query)) + expect_type(prov$codes_summed$observed, "character") + expect_true(all(c("E04") %in% prov$codes_summed$observed)) +}) + +test_that("cog_spending rejects invalid inputs", { + expect_error(cog_spending(123, 2020L), "character") + expect_error(cog_spending("101006006", "2020"), "years") +})