diff --git a/NEWS.md b/NEWS.md index 7c45512..1091972 100644 --- a/NEWS.md +++ b/NEWS.md @@ -19,6 +19,28 @@ at boot -- `getFromNamespace(".ensure_session", "uscogdata")()` followed by a ma `SET threads` -- depending both on a private name and on the session already being open. +## `cog_gov_search()` and `cog_balances()` gain `limit`/`offset` + +Pagination arrived on `cog_spending()`/`cog_revenue()` in 0.3.0; the other two +verbs were left materializing everything and slicing in R. Both now take +`limit`/`offset` with the same semantics: `NULL` default, the page applied in +SQL behind a deterministic `ORDER BY`, and the unpaginated count returned as a +`total_rows` attribute computed by `COUNT(*) OVER()` in the same scan rather +than a second query. + +`cog_gov_search()` had no `LIMIT` at all, which made it the one verb that +returns the entire 40,336-row crosswalk when called with no filter. + +Two refusals rather than silent surprises: + +* `cog_balances(recipe = , limit = )` aborts with class + `uscogdata_recipe_pagination_conflict` -- a recipe's result comes from a + separate query that pagination is not wired into. +* `cog_gov_search()` in basket mode (`length(name) > 1`) aborts with class + `uscogdata_basket_pagination_conflict`. Basket mode returns one resolved row + per requested name with a sidecar covering all of them; a page of that is not + a page of anything the caller asked for. + ## Cohorts can be named by predicate, not just by id `cog_spending()`, `cog_revenue()` and `cog_balances()` gain optional `state` @@ -67,6 +89,14 @@ A `govid`-named cohort's provenance is unchanged. ## Fixes +* `cog_gov_search()` now orders by `population_acs DESC NULLS LAST, + canonical_govid`. **`population_acs` alone is not a total order** -- ties, and + the entire `NULLS LAST` block, came back in whatever order the scan produced. + That was invisible while every call returned the full result set, but it makes + a paged sweep unsound: two requests can order tied rows differently, so a row + is duplicated on one page and missing from the next. Unpaginated results are + unchanged except for the relative order of rows that were already tied. + * An unknown `state` abbreviation now aborts with "Unknown state abbreviation" (class `uscogdata_unknown_state`) instead of base R's "subscript out of bounds". `.state_abbrev_to_fips` is a named character vector, so `[[` on an diff --git a/R/balances.R b/R/balances.R index 26b439c..8ef4e60 100644 --- a/R/balances.R +++ b/R/balances.R @@ -47,6 +47,12 @@ #' @param recipe Optional harmonization recipe id (see [cog_recipes()]). #' `"cash_securities_z77_wide"` and `"cash_securities_z78_wide"` bridge the #' wide era to the modern one. +#' @param limit Maximum number of result rows to return, pushed into the SQL +#' rather than applied after materializing every row. `NULL` (default) +#' returns everything. Cannot be combined with `recipe` -- see `offset` and +#' `total_rows`. +#' @param offset Rows to skip before `limit` starts counting (0-based). +#' Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set. #' #' @return Tibble with columns `year`, `canonical_govid`, `gov_name`, #' `balance_subtype`, `category`, `amt_nominal`, `codes_included`, @@ -64,11 +70,16 @@ #' and `truncated` (the observed subtypes whose coverage falls short of the #' requested years). `expenditure_concept`/`revenue_concept` are `NA` -- #' holdings are a stock, not a flow, so neither concept vocabulary applies. +#' +#' When `limit` is set, also carries a `total_rows` attribute: the full +#' unpaginated row count, computed by the same query (`COUNT(*) OVER()`) +#' rather than a second scan. #' @export cog_balances <- function(govid = NULL, years, category = NULL, per_capita = FALSE, adjust_to_year = NULL, basis = c("harmonized", "raw"), recipe = NULL, - state = NULL, type = NULL) { + state = NULL, type = NULL, + limit = NULL, offset = NULL) { call <- match.call() basis <- match.arg(basis, c("harmonized", "raw")) # Coerce FIRST, validate second: .validate_verb_inputs() asserts @@ -91,6 +102,22 @@ cog_balances <- function(govid = NULL, years, category = NULL, # validator's own doc comment for the incident that made that matter. .validate_verb_inputs(govid, years, category, per_capita, adjust_to_year, recipe) + + # Same semantics as the money verbs (R/pagination.R). Only the `recipe` + # conflict applies here: cog_balances() has no `complete` argument, and a + # recipe's result comes from .run_recipe()'s own query, which pagination is + # not wired into. + paging <- .validate_pagination(limit, offset) + limit <- paging$limit + offset <- paging$offset + if (!is.null(limit) && !is.null(recipe)) { + cli::cli_abort(c( + "`limit`/`offset` cannot be combined with `recipe`.", + "i" = "A recipe's result comes from a separate query (`.run_recipe()`) that pagination is not wired into yet.", + "*" = "Drop `limit`/`offset`, or drop `recipe`." + ), class = "uscogdata_recipe_pagination_conflict") + } + years <- as.integer(years) if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year) @@ -108,6 +135,7 @@ cog_balances <- function(govid = NULL, years, category = NULL, manifest <- .uscogdata_env$manifest recipe_block <- NULL category_for_prov <- category + total_rows <- NULL # set below only when limit is non-NULL (non-recipe path) if (!is.null(recipe)) { .require_schema_v5(con, manifest, "recipe =") @@ -125,8 +153,18 @@ cog_balances <- function(govid = NULL, years, category = NULL, } else { sql <- .build_verb_sql("balance_annotated", "balance_subtype", cohort, years, category, - ig_view = NULL, subtype_scope = NULL) + ig_view = NULL, subtype_scope = NULL, + limit = limit, offset = offset) result <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) + if (!is.null(limit)) { + paged <- .take_pagination_total(result, con, function() { + .build_verb_sql("balance_annotated", "balance_subtype", + cohort, years, category, + ig_view = NULL, subtype_scope = NULL) + }) + result <- paged$result + total_rows <- paged$total_rows + } } # Order matters (matches .verb_spendrev()): per-capita first, so @@ -158,6 +196,7 @@ cog_balances <- function(govid = NULL, years, category = NULL, .emit_balance_caveats(prov$balance_caveats) attr(result, "provenance") <- prov + if (!is.null(limit)) attr(result, "total_rows") <- total_rows result } diff --git a/R/pagination.R b/R/pagination.R new file mode 100644 index 0000000..ae9d4da --- /dev/null +++ b/R/pagination.R @@ -0,0 +1,81 @@ +# R/pagination.R +# +# Shared limit/offset machinery. #39 established the semantics inside +# .verb_spendrev(); #57 extends them to cog_gov_search() and cog_balances(), +# which is what made a single definition worth having: three inline copies of +# "coerce, refuse, unwrap the count" would be three places for the meaning of +# `total_rows` to drift. +# +# The SQL side stays in .build_verb_sql() (R/spending.R) -- it already wraps +# the aggregate in an outer SELECT so COUNT(*) OVER() sees the post-GROUP-BY +# row count rather than the pre-aggregation one, and that is the subtle part +# worth not duplicating either. + +#' Coerce and check a limit/offset pair. +#' +#' Returns the coerced pair, or NULL for `limit` when no page was requested. +#' `offset` defaults to 0 whenever `limit` is set, so a caller can supply just +#' `limit` and get the first page. +#' +#' Conflicts with other arguments are deliberately NOT checked here: they +#' differ per verb (`complete`/`recipe` for the money verbs, basket mode for +#' `cog_gov_search()`, `recipe` alone for `cog_balances()`), and a shared +#' function taking a list of conflict flags would be harder to read than the +#' three explicit refusals at the call sites. +#' @noRd +.validate_pagination <- function(limit, offset) { + if (is.null(limit)) { + return(list(limit = NULL, offset = NULL)) + } + limit <- as.integer(limit) + if (length(limit) != 1L || is.na(limit) || limit < 0L) { + cli::cli_abort("`limit` must be a single non-negative integer.", + class = "uscogdata_invalid_pagination") + } + offset <- if (is.null(offset)) 0L else as.integer(offset) + if (length(offset) != 1L || is.na(offset) || offset < 0L) { + cli::cli_abort("`offset` must be a single non-negative integer.", + class = "uscogdata_invalid_pagination") + } + list(limit = limit, offset = offset) +} + +#' Wrap a query so one page comes back carrying the unpaginated total. +#' +#' `COUNT(*) OVER()` rides along as an ordinary column, so the caller gets the +#' true total from the SAME scan instead of a second round trip. The outer +#' `SELECT *` matters: appending LIMIT/OFFSET directly to a grouped query would +#' have the window function count pre-aggregation rows. +#' @noRd +.paginate_sql <- function(base_sql, limit, offset) { + if (is.null(limit)) return(base_sql) + sprintf( + "SELECT *, COUNT(*) OVER() AS pagination_total_rows + FROM (%s) AS _paged + LIMIT %d OFFSET %d", + base_sql, limit, offset + ) +} + +#' Strip the count column back out and report the unpaginated total. +#' +#' Returns `list(result = , total_rows = )`. +#' +#' An empty page -- an offset past the end -- carries no row to read the window +#' function off, so that one case falls back to a second, unpaginated +#' `COUNT(*)` rather than reporting a wrong zero. `unpaged_sql` is passed as a +#' function so the fallback query is only BUILT when it is actually needed; +#' every caller's unpaginated SQL is otherwise constructed on every paged call +#' and thrown away. +#' @noRd +.take_pagination_total <- function(result, con, unpaged_sql) { + if (nrow(result) > 0L) { + total <- result$pagination_total_rows[[1]] + result$pagination_total_rows <- NULL + return(list(result = result, total_rows = as.integer(total))) + } + count_sql <- sprintf("SELECT COUNT(*) AS n FROM (%s) AS _uncounted", + if (is.function(unpaged_sql)) unpaged_sql() else unpaged_sql) + list(result = result, + total_rows = as.integer(DBI::dbGetQuery(con, count_sql)$n[[1]])) +} diff --git a/R/search.R b/R/search.R index cb3abbd..806e6d8 100644 --- a/R/search.R +++ b/R/search.R @@ -44,10 +44,22 @@ #' in basket mode (recycles from length 1). Excluded types `4`/`5` (or #' `"special_district"` / `"school_district"`) trigger an explanatory #' message and an empty result. +#' @param limit Maximum number of rows to return, applied in SQL. `NULL` +#' (default) returns every match -- which, with no other filter, is the +#' entire crosswalk. Utility mode only: pagination has no meaning in basket +#' mode, where the result is one resolved row per requested name in input +#' order, and is refused there with class +#' `uscogdata_basket_pagination_conflict`. +#' @param offset Rows to skip before `limit` starts counting (0-based). +#' Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set. #' @return A tibble of `canonical_fips_xwalk` rows. In utility mode, all -#' matches sorted by `population_acs` desc. In basket mode, resolved -#' rows in input order, with `attr(., "resolution")` set to the -#' sidecar tibble. +#' matches sorted by `population_acs` desc, ties broken by +#' `canonical_govid`. In basket mode, resolved rows in input order, with +#' `attr(., "resolution")` set to the sidecar tibble. +#' +#' When `limit` is set, carries a `total_rows` attribute: the full +#' unpaginated match count, computed by the same query (`COUNT(*) OVER()`) +#' rather than a second scan. #' @seealso [cog_basket_resolution()], [cog_basket_unresolved()], #' [cog_spending()], [cog_revenue()]. #' @examples @@ -82,7 +94,12 @@ #' ) #' } #' @export -cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { +cog_gov_search <- function(name = NULL, state = NULL, type = NULL, + limit = NULL, offset = NULL) { + paging <- .validate_pagination(limit, offset) + limit <- paging$limit + offset <- paging$offset + if (!is.null(type) && length(type) == 1L && .is_excluded_type(type)) { cli::cli_inform(c( i = "v0.1 covers gov_types 0-3 (state/county/city/township) only.", @@ -94,6 +111,18 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { con <- .ensure_session() if (length(name) > 1L) { + # Basket mode returns one resolved row per requested name, in input order, + # with a resolution sidecar describing how each was matched. A page of that + # is not a page of anything the caller asked for -- the sidecar would still + # describe every name -- so refuse rather than silently ignoring the + # arguments. Same shape as the recipe/complete refusals in .verb_spendrev(). + if (!is.null(limit)) { + cli::cli_abort(c( + "`limit`/`offset` cannot be combined with basket mode.", + "i" = "Basket mode ({.code length(name) > 1}) returns one resolved row per requested name, in input order, with a resolution sidecar covering all of them.", + "*" = "Drop `limit`/`offset`, or search one name at a time." + ), class = "uscogdata_basket_pagination_conflict") + } return(.resolve_basket(name = name, state = state, type = type, con = con)) } @@ -123,12 +152,26 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) { } where <- if (length(preds) == 0L) "" else paste("WHERE", paste(preds, collapse = " AND ")) - sql <- paste( + # canonical_govid breaks ties. population_acs alone is NOT a total order -- + # governments sharing a population, and the whole NULLS LAST block, came back + # in whatever order the scan produced. That was invisible while every call + # returned the full result set, but it makes a paged sweep unsound: two + # requests can order the tied rows differently, so a row is duplicated on one + # page and missing from the next. Any pagination has to sit on a total order. + base_sql <- paste( "SELECT * FROM canonical_fips_xwalk", where, - "ORDER BY population_acs DESC NULLS LAST" + "ORDER BY population_acs DESC NULLS LAST, canonical_govid" ) - tibble::as_tibble(DBI::dbGetQuery(con, sql)) + result <- tibble::as_tibble( + DBI::dbGetQuery(con, .paginate_sql(base_sql, limit, offset)) + ) + if (is.null(limit)) return(result) + + paged <- .take_pagination_total(result, con, base_sql) + out <- paged$result + attr(out, "total_rows") <- paged$total_rows + out } #' @noRd diff --git a/R/spending.R b/R/spending.R index 40407a2..4129a27 100644 --- a/R/spending.R +++ b/R/spending.R @@ -398,17 +398,10 @@ cog_spending <- function(govid = NULL, years, category = NULL, # up front rather than silently ignored: complete = TRUE fills a grid over # the FULL requested (year, category) space, and a recipe's result comes # from .run_recipe()'s own query, which this function does not touch. + paging <- .validate_pagination(limit, offset) + limit <- paging$limit + offset <- paging$offset if (!is.null(limit)) { - limit <- as.integer(limit) - if (length(limit) != 1L || is.na(limit) || limit < 0L) { - cli::cli_abort("`limit` must be a single non-negative integer.", - class = "uscogdata_invalid_pagination") - } - offset <- if (is.null(offset)) 0L else as.integer(offset) - if (length(offset) != 1L || is.na(offset) || offset < 0L) { - cli::cli_abort("`offset` must be a single non-negative integer.", - class = "uscogdata_invalid_pagination") - } if (complete) { cli::cli_abort(c( "`limit`/`offset` cannot be combined with `complete = TRUE`.", @@ -466,24 +459,14 @@ cog_spending <- function(govid = NULL, years, category = NULL, limit = limit, offset = offset) result <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) if (!is.null(limit)) { - # COUNT(*) OVER() rides along as an ordinary column so the total comes - # from the same scan when this page has any rows -- see - # .build_verb_sql(). An empty page (offset past the end) carries no - # such row to read it from, so that one case falls back to a second, - # unpaginated COUNT(*) query rather than reporting a wrong zero. - if (nrow(result) > 0L) { - total_rows <- result$pagination_total_rows[[1]] - result$pagination_total_rows <- NULL - } else { - count_sql <- sprintf( - "SELECT COUNT(*) AS n FROM (%s) AS _uncounted", - .build_verb_sql(view, subtype_col, cohort, years, - if (all_categories) NULL else category, - ig_view, subtype_scope, - all_categories = all_categories) - ) - total_rows <- as.integer(DBI::dbGetQuery(con, count_sql)$n[[1]]) - } + paged <- .take_pagination_total(result, con, function() { + .build_verb_sql(view, subtype_col, cohort, years, + if (all_categories) NULL else category, + ig_view, subtype_scope, + all_categories = all_categories) + }) + result <- paged$result + total_rows <- paged$total_rows } } @@ -873,22 +856,9 @@ cog_spending <- function(govid = NULL, years, category = NULL, # matching row across the network only to slice and discard most of it # afterward (the pattern behind the 2026-08-06 production incident: a # 193,105-row/194-page sweep re-ran the full query and re-listified every - # row on EVERY page). COUNT(*) OVER() rides along as an ordinary column so - # the caller gets the true total from this same scan -- see the call site - # in .verb_spendrev(), which reads it off row 1 and strips it back out. - # The outer SELECT * wrapping (rather than appending LIMIT/OFFSET directly - # to base_sql) is what makes COUNT(*) OVER() see the post-GROUP-BY row - # count, not the pre-aggregation one. - if (is.null(limit)) { - base_sql - } else { - sprintf( - "SELECT *, COUNT(*) OVER() AS pagination_total_rows - FROM (%s) AS _paged - LIMIT %d OFFSET %d", - base_sql, limit, offset - ) - } + # row on EVERY page). See .paginate_sql() in R/pagination.R for why the + # wrapping is an outer SELECT rather than a bare LIMIT on base_sql. + .paginate_sql(base_sql, limit, offset) } #' Join population onto a result and derive the per-capita columns. diff --git a/man/cog_balances.Rd b/man/cog_balances.Rd index e573552..b4ef178 100644 --- a/man/cog_balances.Rd +++ b/man/cog_balances.Rd @@ -13,7 +13,9 @@ cog_balances( basis = c("harmonized", "raw"), recipe = NULL, state = NULL, - type = NULL + type = NULL, + limit = NULL, + offset = NULL ) } \arguments{ @@ -76,6 +78,14 @@ wide era to the modern one.} and `govids_missing` are empty -- there is no id list to report against -- and `provenance$scope$cohort` carries `state`, `type` and `n_governments` instead. A `govid`-named cohort reports exactly as before.} + +\item{limit}{Maximum number of result rows to return, pushed into the SQL +rather than applied after materializing every row. `NULL` (default) +returns everything. Cannot be combined with `recipe` -- see `offset` and +`total_rows`.} + +\item{offset}{Rows to skip before `limit` starts counting (0-based). +Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set.} } \value{ Tibble with columns `year`, `canonical_govid`, `gov_name`, @@ -94,6 +104,10 @@ Tibble with columns `year`, `canonical_govid`, `gov_name`, and `truncated` (the observed subtypes whose coverage falls short of the requested years). `expenditure_concept`/`revenue_concept` are `NA` -- holdings are a stock, not a flow, so neither concept vocabulary applies. + + When `limit` is set, also carries a `total_rows` attribute: the full + unpaginated row count, computed by the same query (`COUNT(*) OVER()`) + rather than a second scan. } \description{ Returns Census cash-and-security holdings (`category_type = "balance"`): diff --git a/man/cog_gov_search.Rd b/man/cog_gov_search.Rd index c596d38..1656b15 100644 --- a/man/cog_gov_search.Rd +++ b/man/cog_gov_search.Rd @@ -4,7 +4,13 @@ \alias{cog_gov_search} \title{Search for governments by name, state, and/or type} \usage{ -cog_gov_search(name = NULL, state = NULL, type = NULL) +cog_gov_search( + name = NULL, + state = NULL, + type = NULL, + limit = NULL, + offset = NULL +) } \arguments{ \item{name}{Character vector of place name(s). Length 1 = utility mode; @@ -19,12 +25,26 @@ all entries; otherwise must match `length(name)`.} in basket mode (recycles from length 1). Excluded types `4`/`5` (or `"special_district"` / `"school_district"`) trigger an explanatory message and an empty result.} + +\item{limit}{Maximum number of rows to return, applied in SQL. `NULL` +(default) returns every match -- which, with no other filter, is the +entire crosswalk. Utility mode only: pagination has no meaning in basket +mode, where the result is one resolved row per requested name in input +order, and is refused there with class +`uscogdata_basket_pagination_conflict`.} + +\item{offset}{Rows to skip before `limit` starts counting (0-based). +Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set.} } \value{ A tibble of `canonical_fips_xwalk` rows. In utility mode, all - matches sorted by `population_acs` desc. In basket mode, resolved - rows in input order, with `attr(., "resolution")` set to the - sidecar tibble. + matches sorted by `population_acs` desc, ties broken by + `canonical_govid`. In basket mode, resolved rows in input order, with + `attr(., "resolution")` set to the sidecar tibble. + + When `limit` is set, carries a `total_rows` attribute: the full + unpaginated match count, computed by the same query (`COUNT(*) OVER()`) + rather than a second scan. } \description{ Resolves human-readable place names into rows of `canonical_fips_xwalk`, diff --git a/tests/testthat/test-search-balances-pagination.R b/tests/testthat/test-search-balances-pagination.R new file mode 100644 index 0000000..b57798d --- /dev/null +++ b/tests/testthat/test-search-balances-pagination.R @@ -0,0 +1,178 @@ +# tests/testthat/test-search-balances-pagination.R +# +# uscogdata#57. cog_spending()/cog_revenue() gained limit/offset in #39; +# cog_gov_search() and cog_balances() did not, so every consumer of those two +# was back to materialize-then-slice -- the exact pattern that wedged the +# production API for hours on 2026-08-06. +# +# cog_gov_search() was also the one verb with no LIMIT at all, so an +# unfiltered call returns the entire 40,336-row crosswalk by accident. + +# --- cog_gov_search() ------------------------------------------------------- + +test_that("cog_gov_search() limit returns the first page of the unpaginated result", { + skip_if_no_corpus() + full <- cog_gov_search(state = "WI", type = "city") + skip_if(nrow(full) < 12L, "fixture has too few WI cities to page") + + page <- cog_gov_search(state = "WI", type = "city", limit = 5L) + expect_equal(nrow(page), 5L) + expect_equal(page$canonical_govid, full$canonical_govid[1:5]) +}) + +test_that("cog_gov_search() offset skips ahead without gaps or overlap", { + skip_if_no_corpus() + full <- cog_gov_search(state = "WI", type = "city") + skip_if(nrow(full) < 12L, "fixture has too few WI cities to page") + + p1 <- cog_gov_search(state = "WI", type = "city", limit = 5L) + p2 <- cog_gov_search(state = "WI", type = "city", limit = 5L, offset = 5L) + expect_equal(p2$canonical_govid, full$canonical_govid[6:10]) + expect_length(intersect(p1$canonical_govid, p2$canonical_govid), 0L) +}) + +test_that("walking every page reconstructs the unpaginated search exactly", { + skip_if_no_corpus() + full <- cog_gov_search(state = "WI", type = "city") + n <- nrow(full) + limit <- 7L + pages <- list() + offset <- 0L + repeat { + p <- cog_gov_search(state = "WI", type = "city", limit = limit, offset = offset) + if (nrow(p) == 0L) break + pages[[length(pages) + 1L]] <- p + offset <- offset + limit + if (offset > n + limit) stop("test runaway: paging did not terminate") + } + walked <- dplyr::bind_rows(pages) + expect_equal(nrow(walked), n) + expect_equal(walked$canonical_govid, full$canonical_govid) +}) + +test_that("cog_gov_search() total_rows reports the full unpaginated count", { + skip_if_no_corpus() + full <- cog_gov_search(state = "WI", type = "city") + page <- cog_gov_search(state = "WI", type = "city", limit = 3L) + expect_equal(attr(page, "total_rows"), nrow(full)) +}) + +test_that("cog_gov_search() offset past the end reports the true total, not zero", { + skip_if_no_corpus() + full <- cog_gov_search(state = "WI", type = "city") + # No row survives to carry COUNT(*) OVER(), so this is the branch that has + # to fall back to a second count rather than reporting 0 rows out of 0. + page <- cog_gov_search(state = "WI", type = "city", + limit = 5L, offset = nrow(full) + 50L) + expect_equal(nrow(page), 0L) + expect_equal(attr(page, "total_rows"), nrow(full)) +}) + +test_that("cog_gov_search() bounds an otherwise-unfiltered crosswalk sweep", { + skip_if_no_corpus() + # The reason this verb needed a limit most: with no filter it returns the + # whole crosswalk. + page <- cog_gov_search(limit = 10L) + expect_equal(nrow(page), 10L) + expect_gt(attr(page, "total_rows"), 10L) +}) + +test_that("cog_gov_search() orders by a total order, not population alone", { + skip_if_no_corpus() + # population_acs is not unique -- NA in particular repeats across many rows + # -- so paging on it alone can duplicate a row on one page and drop it from + # the next. The tiebreaker is what makes the sequence reproducible. + full <- cog_gov_search(state = "WI") + skip_if(nrow(full) < 5L, "fixture has too few WI governments") + expect_equal(cog_gov_search(state = "WI")$canonical_govid, + full$canonical_govid) + + ties <- full[is.na(full$population_acs), ] + skip_if(nrow(ties) < 2L, "no tied rows in the fixture to order") + expect_false(is.unsorted(ties$canonical_govid)) +}) + +test_that("cog_gov_search() refuses pagination in basket mode", { + skip_if_no_corpus() + expect_error( + cog_gov_search(name = c("MADISON CITY", "MILWAUKEE CITY"), + state = c("WI", "WI"), limit = 1L), + class = "uscogdata_basket_pagination_conflict" + ) +}) + +test_that("cog_gov_search() rejects a malformed limit or offset", { + skip_if_no_corpus() + expect_error(cog_gov_search(state = "WI", limit = -1L), + class = "uscogdata_invalid_pagination") + expect_error(cog_gov_search(state = "WI", limit = 5L, offset = -1L), + class = "uscogdata_invalid_pagination") +}) + +# --- cog_balances() --------------------------------------------------------- + +test_that("cog_balances() limit/offset walk the unpaginated result exactly", { + skip_if_no_corpus() + full <- cog_balances(years = 2019:2020, state = "WI", type = "city") + skip_if(nrow(full) < 6L, "fixture has too few WI city balance rows to page") + + key <- c("year", "canonical_govid", "balance_subtype", "amt_nominal") + p1 <- cog_balances(years = 2019:2020, state = "WI", type = "city", limit = 3L) + p2 <- cog_balances(years = 2019:2020, state = "WI", type = "city", + limit = 3L, offset = 3L) + + expect_equal(nrow(p1), 3L) + expect_equal(p1[key], full[1:3, key], ignore_attr = TRUE) + expect_equal(p2[key], full[4:6, key], ignore_attr = TRUE) + # The window-function column is an implementation detail and must not reach + # the caller's data frame. + expect_false("pagination_total_rows" %in% names(p1)) +}) + +test_that("cog_balances() total_rows reports the full unpaginated count", { + skip_if_no_corpus() + full <- cog_balances(years = 2019:2020, state = "WI", type = "city") + page <- cog_balances(years = 2019:2020, state = "WI", type = "city", limit = 2L) + expect_equal(attr(page, "total_rows"), nrow(full)) +}) + +test_that("cog_balances() offset past the end reports the true total", { + skip_if_no_corpus() + full <- cog_balances(years = 2019:2020, state = "WI", type = "city") + page <- cog_balances(years = 2019:2020, state = "WI", type = "city", + limit = 5L, offset = nrow(full) + 50L) + expect_equal(nrow(page), 0L) + expect_equal(attr(page, "total_rows"), nrow(full)) +}) + +test_that("cog_balances() refuses pagination alongside a recipe", { + skip_if_no_corpus() + expect_error( + cog_balances(years = 2011, state = "WI", type = "city", + recipe = "cash_securities_z77_wide", limit = 5L), + class = "uscogdata_recipe_pagination_conflict" + ) +}) + +test_that("cog_balances() rejects a malformed limit or offset", { + skip_if_no_corpus() + expect_error(cog_balances(years = 2019, state = "WI", type = "city", limit = -1L), + class = "uscogdata_invalid_pagination") + expect_error(cog_balances(years = 2019, state = "WI", type = "city", + limit = 5L, offset = -1L), + class = "uscogdata_invalid_pagination") +}) + +# --- Unchanged without the arguments ---------------------------------------- + +test_that("both verbs are unchanged when limit is not supplied", { + skip_if_no_corpus() + # The adoption contract for cog-api: NULL default, so a formals() probe can + # feature-detect without any call site changing behaviour. + s <- cog_gov_search(state = "WI", type = "city") + b <- cog_balances(years = 2019, state = "WI", type = "city") + expect_null(attr(s, "total_rows")) + expect_null(attr(b, "total_rows")) + expect_true(all(c("limit", "offset") %in% names(formals(cog_gov_search)))) + expect_true(all(c("limit", "offset") %in% names(formals(cog_balances)))) +})