Compare commits
17
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d0d724c4c8
|
||
|
|
56f610ea3f | ||
|
|
62741343ee | ||
|
|
392643bd74
|
||
|
|
0c7c7eb299
|
||
|
|
7274ce3bfe
|
||
|
|
24e86ed598
|
||
|
|
1ec20174b7
|
||
|
|
2e317d3a0f
|
||
|
|
f7c606984f | ||
|
|
9617b86a26
|
||
|
|
224e5d0530 | ||
|
|
5f81ae386b
|
||
|
|
d2caa6de97 | ||
|
|
303aa59b07
|
||
|
|
81f72321ee
|
||
|
|
5cb83d8f1d |
@@ -3,6 +3,7 @@
|
||||
^\.Rproj\.user$
|
||||
^_pkgdown\.yml$
|
||||
^docs$
|
||||
^pm$
|
||||
^Meta$
|
||||
^doc$
|
||||
^pkgdown$
|
||||
|
||||
@@ -39,5 +39,17 @@ jobs:
|
||||
echo "PAT_GH is unset -- add it under Settings > Actions > Secrets." >&2
|
||||
exit 1
|
||||
fi
|
||||
# On a tag-triggered run, checkout materializes refs/tags/<tag> as a
|
||||
# LIGHTWEIGHT tag at the commit SHA -- the annotated tag object Gitea
|
||||
# holds is never fetched. Mirroring that strips the annotation, and the
|
||||
# NEXT run on main (which does fetch the real object) is then rejected
|
||||
# with "already exists" trying to correct it, because git will not
|
||||
# clobber an existing tag. That is why v0.4.0 failed to mirror.
|
||||
#
|
||||
# Re-fetch canonical tag objects from Gitea first. --force here rewrites
|
||||
# LOCAL tag refs only; it is not a force push and does not weaken the
|
||||
# non-force guarantee on main documented above.
|
||||
git fetch --tags --force origin
|
||||
|
||||
git push "https://x-access-token:${PAT_GH}@github.com/civilytics/uscogdata.git" \
|
||||
HEAD:refs/heads/main --tags
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
# Explain the mirror contribution flow on every incoming pull request.
|
||||
#
|
||||
# This repository is a MIRROR. A PR opened here is landed on the canonical Gitea
|
||||
# repository and syncs back; because the merge preserves the contributor's
|
||||
# commits at their original SHAs, GitHub marks the PR "Merged" on its own as
|
||||
# soon as the mirror syncs -- with nobody visibly clicking Merge.
|
||||
#
|
||||
# Without this comment, that reads as a rejection: the contributor sees their PR
|
||||
# close with no review, no merge button pressed, and no explanation. It is
|
||||
# actually the successful outcome. Say so up front, before it happens.
|
||||
#
|
||||
# WHY pull_request_target AND NOT pull_request:
|
||||
# a `pull_request` run from a fork gets a read-only token, so it cannot post a
|
||||
# comment -- which is exactly the case this workflow exists to serve.
|
||||
# `pull_request_target` runs in the context of the BASE repo and gets a writable
|
||||
# token. That is only safe because this job never checks out or executes the
|
||||
# contributor's code; it posts a fixed string. Do not add a checkout of
|
||||
# `github.event.pull_request.head.sha` here -- that combination is the standard
|
||||
# pull_request_target privilege-escalation hole.
|
||||
name: Explain the mirror flow
|
||||
|
||||
on:
|
||||
pull_request_target:
|
||||
types: [opened]
|
||||
|
||||
permissions:
|
||||
pull-requests: write
|
||||
|
||||
jobs:
|
||||
comment:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Post the contribution-flow explainer
|
||||
uses: actions/github-script@v7
|
||||
with:
|
||||
script: |
|
||||
const body = [
|
||||
"Thanks for this — and one thing worth knowing before it happens.",
|
||||
"",
|
||||
"**This repository is a mirror.** Development happens on Gitea at",
|
||||
"`gitea.civilytics.org/Civilytics/uscogdata`. Your pull request will be fetched",
|
||||
"from here, landed there, and synced back.",
|
||||
"",
|
||||
"Because that merge preserves your commits at their original SHAs, **GitHub will",
|
||||
"mark this pull request \"Merged\" on its own** — without anyone visibly clicking",
|
||||
"the Merge button, and possibly without a review comment on this page first.",
|
||||
"",
|
||||
"> If your pull request closes as \"Merged\" and nobody appears to have merged it,",
|
||||
"> that is the normal, successful outcome — not a rejection.",
|
||||
"",
|
||||
"If it is *not* going to be merged, you will get an actual reply saying so.",
|
||||
"",
|
||||
"Substantial contributions get a `ctb` entry in `DESCRIPTION`, which surfaces in",
|
||||
"`citation(\"uscogdata\")`. There is no CLA and no DCO sign-off.",
|
||||
"",
|
||||
"Full details: [CONTRIBUTING.md](https://github.com/civilytics/uscogdata/blob/main/CONTRIBUTING.md).",
|
||||
].join("\n");
|
||||
|
||||
await github.rest.issues.createComment({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
issue_number: context.payload.pull_request.number,
|
||||
body,
|
||||
});
|
||||
@@ -4,6 +4,10 @@
|
||||
.Ruserdata
|
||||
*.Rproj
|
||||
inst/doc
|
||||
# pkgdown output. Compass used to keep its files in docs/pm/ and
|
||||
# docs/decisions/, which forced this to be written as children with two
|
||||
# re-includes -- git cannot re-include anything beneath an excluded directory.
|
||||
# Compass lives in pm/ now, so the whole directory can be excluded again.
|
||||
docs/
|
||||
/doc/
|
||||
/Meta/
|
||||
@@ -12,3 +16,6 @@ docs/
|
||||
|
||||
# SDD working artifacts (ledger, briefs, review packages) — plans/ stays tracked
|
||||
.superpowers/sdd/
|
||||
.compass-cache/
|
||||
# roborev snapshots
|
||||
/.roborev/
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
# roborev configuration, initialised by compass.
|
||||
# Reviews are queued to a background daemon -- they never block a commit.
|
||||
|
||||
post_commit_review = 'commit'
|
||||
excluded_commit_patterns = ['WIP', 'chore:', 'chore(', 'docs:', 'Merge ']
|
||||
|
||||
review_guidelines = '''
|
||||
# --- compass:begin (generated -- edit the sources, not this) ---
|
||||
- Prefer returning new values to mutating arguments in place. A function that edits
|
||||
its caller's object is a bug waiting for a second caller.
|
||||
- Validate at system boundaries -- user input, API responses, file contents, config.
|
||||
Fail fast with a message naming the field and the file.
|
||||
- Never swallow an error. Handle it or let it propagate; a bare catch that continues
|
||||
is worse than a crash.
|
||||
- No hardcoded secrets, tokens, or credentials, and no secrets in log output or error
|
||||
messages.
|
||||
- Parameterise every query. String-built SQL is a defect even when the input looks safe.
|
||||
- Keep functions under roughly 50 lines and files under roughly 400. Flag nesting
|
||||
deeper than four levels.
|
||||
- No magic numbers or hardcoded paths -- name them as constants or read them from config.
|
||||
- New behaviour needs a test. A bug fix needs a test that fails without the fix.
|
||||
- Prose a person reads -- an issue title or body, a journal entry, a decision record,
|
||||
the narrative on the status board -- names the action or the thing, not the shape of
|
||||
the machinery. Flag "gate", "seam", "surface area", "load-bearing", "first-class",
|
||||
"primitive", "blast radius". A project's own defined vocabulary is not the target.
|
||||
- Use the native pipe `|>`, not magrittr `%>%`.
|
||||
- snake_case for objects and functions; UPPER_SNAKE for constants. Never use `.` as a
|
||||
word separator in a function name -- it collides with S3 dispatch.
|
||||
- Validate arguments at the top of exported functions with `stopifnot()` or an explicit
|
||||
check, and say which argument was wrong.
|
||||
- Never `setDT()`, `set()`, or otherwise modify by reference a data.table the caller
|
||||
still owns. `as.data.table()` copies; use it.
|
||||
- Prefer `vapply()` to `sapply()` -- `sapply()` silently returns a list when the type
|
||||
varies, which turns a type error into a downstream mystery.
|
||||
- Use `seq_len(n)` / `seq_along(x)`, never `1:n`, which iterates backwards when n is 0.
|
||||
- Compare strings with `==` only after checking for NA; use `identical()` for scalars
|
||||
where NA would be wrong.
|
||||
- Do not call `library()` inside package or module files; attach packages in scripts and
|
||||
test helpers only.
|
||||
- Namespace-qualify calls into other packages (`stats::sd`) in code that is sourced.
|
||||
- Every exported function needs roxygen with `@param` for each argument (type, meaning,
|
||||
and why the default is what it is) and `@return`. Add `@examples` for exported API.
|
||||
- Declare dependencies in DESCRIPTION. Prefer base R or an existing dependency over
|
||||
adding a new one; a package with zero hard deps is worth keeping that way.
|
||||
- Signal errors with `stop()` carrying a condition class, so callers can catch the kind
|
||||
rather than matching on message text.
|
||||
- Keep internals internal. Export only what a user needs; an accidentally exported
|
||||
helper becomes an API you have to keep.
|
||||
- Tests use testthat edition 3. Each test is self-sufficient -- no reliance on state
|
||||
left by an earlier test or on a fixture built elsewhere in the file.
|
||||
- Prefer duplication in tests over a helper that hides what is being asserted.
|
||||
- Every verb calls .ensure_session() first, then queries via DBI::dbGetQuery().
|
||||
- A verb's return value is always a tbl_df carrying a provenance attribute.
|
||||
- govid inputs always go through .coerce_govid_input(); it accepts a character vector or a data frame.
|
||||
- SQL has two layers: view definitions are numbered .sql files in inst/sql/ registered by .register_views(); query construction is inline sprintf() in R. Add a view as a file; build a query in R.
|
||||
- No arrow dependency -- DuckDB reads parquet natively.
|
||||
- withr is Suggests-only and must appear in tests alone.
|
||||
- Tests must pass offline against the bundled fixture; tests/testthat/setup.R sets USCOGDATA_URL for that.
|
||||
# --- compass:end ---
|
||||
'''
|
||||
+10
-2
@@ -101,5 +101,13 @@ but "usually" is not a release gate.
|
||||
variables set. This is the only check that catches a
|
||||
corpus-unreachable defect, and its absence is why 0.3.0 needed fixing.
|
||||
8. Bump `Version` and add a `NEWS.md` section.
|
||||
9. Tag, then update the r-universe registry pin at
|
||||
`github.com/civilytics/civilytics.r-universe.dev`.
|
||||
9. Tag on **Gitea** (`git tag -a vX.Y.Z && git push origin vX.Y.Z`). The mirror
|
||||
workflow carries tags to GitHub on its own — confirm the tag appears at
|
||||
`github.com/civilytics/uscogdata/tags` before continuing.
|
||||
10. Update the r-universe registry pin at
|
||||
`github.com/civilytics/civilytics.r-universe.dev` — edit `packages.json`'s
|
||||
`branch` to the new tag. **r-universe will not pick up a release until this
|
||||
is edited**: the pin is a tag, deliberately, so a mid-refactor `main` is
|
||||
never published as a release. `"branch": "*release"` would track releases
|
||||
automatically, but it needs a GitHub *Release* object and the mirror pushes
|
||||
tags only — so it would silently never update.
|
||||
|
||||
@@ -1,46 +1,5 @@
|
||||
# uscogdata 0.4.0
|
||||
|
||||
## DuckDB's resource budget is configurable
|
||||
|
||||
`USCOGDATA_DUCKDB_THREADS` and `USCOGDATA_DUCKDB_MEMORY_LIMIT` (with matching
|
||||
`options(uscogdata.duckdb_threads = )` / `options(uscogdata.duckdb_memory_limit = )`
|
||||
spellings) cap the DuckDB connection the package opens. Both follow the same
|
||||
env-var > option > default precedence as `USCOGDATA_URL`.
|
||||
|
||||
Unset, **no pragma is issued at all** and DuckDB's own defaults apply exactly as
|
||||
before -- every visible core. That is right for one interactive session on a
|
||||
dedicated machine and wrong for a server: where several readers share a host, each
|
||||
otherwise claims the whole machine and they contend. Capping measured ~5% on a
|
||||
single-government all-years query (502 ms at 2 threads vs 475 ms uncapped on 16
|
||||
cores), which is cheap enough that a server should always cap.
|
||||
|
||||
This replaces a workaround in which a consumer reached into the package namespace
|
||||
at boot -- `getFromNamespace(".ensure_session", "uscogdata")()` followed by a manual
|
||||
`SET threads` -- depending both on a private name and on the session already being
|
||||
open.
|
||||
|
||||
## `cog_gov_search()` and `cog_balances()` gain `limit`/`offset`
|
||||
|
||||
Pagination arrived on `cog_spending()`/`cog_revenue()` in 0.3.0; the other two
|
||||
verbs were left materializing everything and slicing in R. Both now take
|
||||
`limit`/`offset` with the same semantics: `NULL` default, the page applied in
|
||||
SQL behind a deterministic `ORDER BY`, and the unpaginated count returned as a
|
||||
`total_rows` attribute computed by `COUNT(*) OVER()` in the same scan rather
|
||||
than a second query.
|
||||
|
||||
`cog_gov_search()` had no `LIMIT` at all, which made it the one verb that
|
||||
returns the entire 40,336-row crosswalk when called with no filter.
|
||||
|
||||
Two refusals rather than silent surprises:
|
||||
|
||||
* `cog_balances(recipe = , limit = )` aborts with class
|
||||
`uscogdata_recipe_pagination_conflict` -- a recipe's result comes from a
|
||||
separate query that pagination is not wired into.
|
||||
* `cog_gov_search()` in basket mode (`length(name) > 1`) aborts with class
|
||||
`uscogdata_basket_pagination_conflict`. Basket mode returns one resolved row
|
||||
per requested name with a sidecar covering all of them; a page of that is not
|
||||
a page of anything the caller asked for.
|
||||
|
||||
## Cohorts can be named by predicate, not just by id
|
||||
|
||||
`cog_spending()`, `cog_revenue()` and `cog_balances()` gain optional `state`
|
||||
@@ -87,6 +46,71 @@ When the cohort is named by predicate there is no id list to report, so
|
||||
`provenance$scope$cohort` carries `state`, `type` and `n_governments` instead.
|
||||
A `govid`-named cohort's provenance is unchanged.
|
||||
|
||||
## `cog_gov_search()` and `cog_balances()` gain `limit`/`offset`
|
||||
|
||||
Pagination arrived on `cog_spending()`/`cog_revenue()` in 0.3.0; the other two
|
||||
verbs were left materializing everything and slicing in R. Both now take
|
||||
`limit`/`offset` with the same semantics: `NULL` default, the page applied in
|
||||
SQL behind a deterministic `ORDER BY`, and the unpaginated count returned as a
|
||||
`total_rows` attribute computed by `COUNT(*) OVER()` in the same scan rather
|
||||
than a second query.
|
||||
|
||||
`cog_gov_search()` had no `LIMIT` at all, which made it the one verb that
|
||||
returns the entire 40,336-row crosswalk when called with no filter.
|
||||
|
||||
Two refusals rather than silent surprises:
|
||||
|
||||
* `cog_balances(recipe = , limit = )` aborts with class
|
||||
`uscogdata_recipe_pagination_conflict` -- a recipe's result comes from a
|
||||
separate query that pagination is not wired into.
|
||||
* `cog_gov_search()` in basket mode (`length(name) > 1`) aborts with class
|
||||
`uscogdata_basket_pagination_conflict`. Basket mode returns one resolved row
|
||||
per requested name with a sidecar covering all of them; a page of that is not
|
||||
a page of anything the caller asked for.
|
||||
|
||||
## DuckDB's resource budget is configurable
|
||||
|
||||
`USCOGDATA_DUCKDB_THREADS` and `USCOGDATA_DUCKDB_MEMORY_LIMIT` (with matching
|
||||
`options(uscogdata.duckdb_threads = )` / `options(uscogdata.duckdb_memory_limit = )`
|
||||
spellings) cap the DuckDB connection the package opens. Both follow the same
|
||||
env-var > option > default precedence as `USCOGDATA_URL`.
|
||||
|
||||
Unset, **no pragma is issued at all** and DuckDB's own defaults apply exactly as
|
||||
before -- every visible core. That is right for one interactive session on a
|
||||
dedicated machine and wrong for a server: where several readers share a host, each
|
||||
otherwise claims the whole machine and they contend. Capping measured ~5% on a
|
||||
single-government all-years query (502 ms at 2 threads vs 475 ms uncapped on 16
|
||||
cores), which is cheap enough that a server should always cap.
|
||||
|
||||
This replaces a workaround in which a consumer reached into the package namespace
|
||||
at boot -- `getFromNamespace(".ensure_session", "uscogdata")()` followed by a manual
|
||||
`SET threads` -- depending both on a private name and on the session already being
|
||||
open.
|
||||
|
||||
## Documentation: the corpus-access table is re-measured and honest
|
||||
|
||||
The README's "two ways to read the corpus" table carried figures taken before
|
||||
the corpus was re-chunked into row groups (cog_pipeline#93, published
|
||||
2026-08-09) and reported the mirrored column as "local speed" with no number at
|
||||
all. Re-measured 2026-08-10 against the published corpus (`pipeline_commit
|
||||
3d28ddd`), fresh R session per arm:
|
||||
|
||||
* **A local mirror is roughly 60-80x faster.** A one-off question costs ~12 s
|
||||
end to end remotely against ~0.15 s mirrored. That is the largest single
|
||||
difference available to a user and it is now stated outright rather than left
|
||||
as "local speed".
|
||||
* **Opening the session is the largest remote cost** (~7.5 s -- manifest fetch
|
||||
plus 23 view registrations over HTTPS), larger than any individual query, and
|
||||
it lands on the first query rather than on `library(uscogdata)`. The old table
|
||||
did not account for it anywhere.
|
||||
* **The remote cost is round-trips, not scanning.** A repeat query over
|
||||
already-touched partitions is ~1.5 s against ~4 s cold, and a full-history
|
||||
query costs ~7 s whether it runs first or last.
|
||||
* The corpus size is **~201 MB**, not 190.6 MB -- row-group chunking added ~3.4%
|
||||
and the old figure was ambiguous between MB and MiB besides.
|
||||
* Documented that a burst of remote queries can be rate-limited by the host
|
||||
(`HTTP 429`), which is another reason to mirror for real work.
|
||||
|
||||
## Fixes
|
||||
|
||||
* `cog_gov_search()` now orders by `population_acs DESC NULLS LAST,
|
||||
|
||||
+168
-54
@@ -84,6 +84,21 @@
|
||||
#' @return List of `list(recipe_id, label, available_years, hint,
|
||||
#' ig_recipe_id, trigger, suppressed_amount, suppressed_years,
|
||||
#' suppressed_codes)`, possibly empty.
|
||||
#'
|
||||
#' Decomposed (Issue #33) into three extracted helpers to stay within the
|
||||
#' project's "functions under 50 lines" convention:
|
||||
#' \itemize{
|
||||
#' \item `.query_candidate_recipes()` -- candidate recipe lookup by
|
||||
#' category/subtype scope + `category_type` filter (#34) + M/L exclusion.
|
||||
#' \item `.query_recipe_meta()` -- metadata (label, year spans).
|
||||
#' \item `.query_covered_years()` -- Path 1 gap-year coverage via the
|
||||
#' recipe's own generic join.
|
||||
#' }
|
||||
#' The for-loop that merges covered-years + suppressed-components into
|
||||
#' suggestion objects stays inline here because it interleaves
|
||||
#' empty_hit/supp_hit precedence with field assembly. Likewise kept inline:
|
||||
#' the M/L-exclusion design-comment block and the final
|
||||
#' `.attach_ig_counterparts()` call.
|
||||
#' @noRd
|
||||
.build_suggestions <- function(con, cohort, years, category, result, basis,
|
||||
flow_prefixes, long_view,
|
||||
@@ -111,28 +126,16 @@
|
||||
# by `category` (`.ALL_CATEGORIES` is never a row in
|
||||
# `summary_categories.category`, so a category-keyed sub-select always
|
||||
# came back empty here). The M/L exclusion below is unchanged either way.
|
||||
candidate_scope_sql <- if (isTRUE(all_categories)) {
|
||||
sprintf(
|
||||
"SELECT DISTINCT item_code FROM summary_categories WHERE %s IN (%s)",
|
||||
subtype_col, .sql_lit_chr(subtype_scope)
|
||||
)
|
||||
} else {
|
||||
sprintf(
|
||||
"SELECT DISTINCT item_code FROM summary_categories WHERE category IN (%s)",
|
||||
.sql_lit_chr(category)
|
||||
)
|
||||
}
|
||||
candidates <- DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||
WHERE component_code IN (
|
||||
%s
|
||||
)
|
||||
AND recipe_id NOT IN (
|
||||
SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||
WHERE LEFT(component_code, 1) IN ('M', 'L')
|
||||
)",
|
||||
candidate_scope_sql
|
||||
))$recipe_id
|
||||
#
|
||||
# Issue #34: scope the candidate query by `category_type` ('expenditure'
|
||||
# vs 'revenue') to prevent cross-flow-family leakage -- e.g.
|
||||
# `cog_revenue(category = "Corrections")` must not surface
|
||||
# expenditure-only recipes (E04/E05) merely because they share the same
|
||||
# category name in summary_categories. The type is derived from
|
||||
# flow_prefixes: E/F/G -> 'expenditure', anything else -> 'revenue'.
|
||||
candidates <- .query_candidate_recipes(con, category, flow_prefixes,
|
||||
all_categories, subtype_col,
|
||||
subtype_scope)
|
||||
if (length(candidates) == 0L) return(list())
|
||||
|
||||
result_years <- if (is.null(result) || nrow(result) == 0L) {
|
||||
@@ -164,36 +167,11 @@
|
||||
|
||||
if (length(gap_years) == 0L && nrow(supp) == 0L) return(list())
|
||||
|
||||
meta <- tibble::as_tibble(DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT recipe_id, any_value(label) AS label,
|
||||
MIN(year_min) AS year_min, MAX(year_max) AS year_max
|
||||
FROM harmonization_recipes
|
||||
WHERE recipe_id IN (%s)
|
||||
GROUP BY recipe_id",
|
||||
.sql_lit_chr(candidates)
|
||||
)))
|
||||
meta <- .query_recipe_meta(con, candidates)
|
||||
|
||||
# Path 1 (unchanged): (recipe, year) pairs the recipe's own generic join
|
||||
# covers for this government, restricted to the gap years.
|
||||
covered <- if (length(gap_years) == 0L) {
|
||||
data.frame(recipe_id = character(0), year = integer(0))
|
||||
} else {
|
||||
DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT DISTINCT r.recipe_id, l.year
|
||||
FROM long l
|
||||
JOIN harmonization_recipes r
|
||||
ON l.item_code = r.component_code
|
||||
AND l.year BETWEEN r.year_min AND r.year_max
|
||||
AND (r.gov_type_scope = 'all'
|
||||
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||
WHERE r.recipe_id IN (%s)
|
||||
AND %s
|
||||
AND l.year IN (%s)",
|
||||
.sql_lit_chr(candidates), .cohort_sql(cohort, "l.canonical_govid"),
|
||||
paste(gap_years, collapse = ",")
|
||||
))
|
||||
}
|
||||
covered <- .query_covered_years(con, candidates, cohort, gap_years)
|
||||
|
||||
suggestions <- list()
|
||||
for (rid in candidates) {
|
||||
@@ -227,6 +205,146 @@
|
||||
.attach_ig_counterparts(con, suggestions, flow_prefixes)
|
||||
}
|
||||
|
||||
#' Query candidate harmonization recipe IDs for a coverage-gap suggestion.
|
||||
#'
|
||||
#' Selects recipes whose component codes fall within the requested scope
|
||||
#' (category or subtype allowlist), excluding any recipe that is ITSELF an
|
||||
#' intergovernmental (M/L) recipe -- i.e. every one of its own component
|
||||
#' codes is M/L-prefixed. Without this exclusion, a category whose
|
||||
#' summary_categories rows span both a Direct family (e.g. E04/E05,
|
||||
#' "Corrections") and its M/L counterpart (M04/M05) makes the M/L recipe
|
||||
#' itself a raw top-level candidate for a plain `cog_spending()` call --
|
||||
#' following that hint would silently return intergovernmental dollars
|
||||
#' under `expenditure_concept = "direct"` provenance.
|
||||
#'
|
||||
#' In all-categories mode (`all_categories = TRUE`) the inner sub-select is
|
||||
#' scoped by `subtype_col`/`subtype_scope` -- the same allowlist
|
||||
#' `.build_verb_sql()` applies as a WHERE predicate to make the summed
|
||||
#' result a *concept* (see R/spending.R), not by `category`.
|
||||
#' `.ALL_CATEGORIES` ("All Categories") is never itself a row in
|
||||
#' `summary_categories.category`, so a category-keyed sub-select always
|
||||
#' returns zero candidates and silently disables signposting.
|
||||
#'
|
||||
#' Scope is also by `category_type` ('expenditure' vs 'revenue', Issue #34)
|
||||
#' to prevent cross-flow-family leakage: `cog_revenue(category =
|
||||
#' "Corrections")` must not surface expenditure-only recipes (E04/E05)
|
||||
#' merely because they share the same category name in summary_categories.
|
||||
#' The type is derived from flow_prefixes: E/F/G -> 'expenditure', anything
|
||||
#' else -> 'revenue'.
|
||||
#'
|
||||
#' @param con Active DuckDB connection.
|
||||
#' @param category Category name, or `NULL`.
|
||||
#' @param flow_prefixes The calling verb's own flow-type prefixes (see
|
||||
#' `.build_suggestions()`). Used to derive `category_type` (#34).
|
||||
#' @param all_categories `TRUE` when the caller used `.ALL_CATEGORIES`.
|
||||
#' @param subtype_col Name of the summary_categories subtype column to
|
||||
#' scope by when `all_categories = TRUE`; ignored otherwise.
|
||||
#' @param subtype_scope Character vector of subtype values to scope by
|
||||
#' when `all_categories = TRUE`; ignored otherwise.
|
||||
#' @return Character vector of recipe IDs (possibly empty).
|
||||
#' @noRd
|
||||
.query_candidate_recipes <- function(con, category, flow_prefixes,
|
||||
all_categories = FALSE,
|
||||
subtype_col = NULL,
|
||||
subtype_scope = NULL) {
|
||||
# Issue #34: derive category_type from flow_prefixes to prevent
|
||||
# cross-flow-family leakage -- e.g. cog_revenue(category = "Corrections")
|
||||
# must not surface expenditure-only recipes merely because they share the
|
||||
# same category name in summary_categories.
|
||||
category_type <- if (all(flow_prefixes %in% c("E", "F", "G"))) {
|
||||
"expenditure"
|
||||
} else {
|
||||
"revenue"
|
||||
}
|
||||
|
||||
candidate_scope_sql <- if (isTRUE(all_categories)) {
|
||||
sprintf(
|
||||
"SELECT DISTINCT item_code FROM summary_categories
|
||||
WHERE %s IN (%s) AND category_type = '%s'",
|
||||
subtype_col, .sql_lit_chr(subtype_scope), category_type
|
||||
)
|
||||
} else {
|
||||
sprintf(
|
||||
"SELECT DISTINCT item_code FROM summary_categories
|
||||
WHERE category IN (%s) AND category_type = '%s'",
|
||||
.sql_lit_chr(category), category_type
|
||||
)
|
||||
}
|
||||
|
||||
DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||
WHERE component_code IN (
|
||||
%s
|
||||
)
|
||||
AND recipe_id NOT IN (
|
||||
SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||
WHERE LEFT(component_code, 1) IN ('M', 'L')
|
||||
)",
|
||||
candidate_scope_sql
|
||||
))$recipe_id
|
||||
}
|
||||
|
||||
#' Query gap-year coverage: which (recipe, year) pairs the recipe's own
|
||||
#' generic join covers for this government, restricted to `gap_years`.
|
||||
#'
|
||||
#' This is Path 1 of a suggestion (unchanged): it finds recipes whose
|
||||
#' component codes' generic join produces at least one row for this
|
||||
#' government in each gap year -- i.e. the category returned nothing in
|
||||
#' that year but a recipe would fill it.
|
||||
#'
|
||||
#' @param con Active DuckDB connection.
|
||||
#' @param candidates Character vector of recipe IDs to check coverage for.
|
||||
#' @param cohort The verb's cohort object (see `.make_cohort()`), rendered
|
||||
#' into the govid predicate on the joined `long` scan via `.cohort_sql()`.
|
||||
#' @param gap_years Integer vector of requested years absent from the
|
||||
#' result.
|
||||
#' @return Data frame with columns `recipe_id` (character) and `year`
|
||||
#' (integer). Returns an empty data frame (`recipe_id = character(0)`,
|
||||
#' `year = integer(0)`) when `gap_years` or `candidates` is empty, so
|
||||
#' callers can safely reference `$recipe_id`.
|
||||
#' @noRd
|
||||
.query_covered_years <- function(con, candidates, cohort, gap_years) {
|
||||
if (length(gap_years) == 0L || length(candidates) == 0L) {
|
||||
return(data.frame(recipe_id = character(0), year = integer(0)))
|
||||
}
|
||||
res <- DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT DISTINCT r.recipe_id, l.year
|
||||
FROM long l
|
||||
JOIN harmonization_recipes r
|
||||
ON l.item_code = r.component_code
|
||||
AND l.year BETWEEN r.year_min AND r.year_max
|
||||
AND (r.gov_type_scope = 'all'
|
||||
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||
WHERE r.recipe_id IN (%s)
|
||||
AND %s
|
||||
AND l.year IN (%s)",
|
||||
.sql_lit_chr(candidates), .cohort_sql(cohort, "l.canonical_govid"),
|
||||
paste(gap_years, collapse = ",")
|
||||
))
|
||||
res$year <- as.integer(res$year)
|
||||
res
|
||||
}
|
||||
|
||||
#' Query recipe metadata: labels and year spans for a set of candidate
|
||||
#' recipes.
|
||||
#'
|
||||
#' @param con Active DuckDB connection.
|
||||
#' @param candidates Character vector of recipe IDs to look up.
|
||||
#' @return Tibble with columns `recipe_id`, `label`, `year_min` (int), and
|
||||
#' `year_max` (int).
|
||||
#' @noRd
|
||||
.query_recipe_meta <- function(con, candidates) {
|
||||
tibble::as_tibble(DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT recipe_id, any_value(label) AS label,
|
||||
MIN(year_min) AS year_min, MAX(year_max) AS year_max
|
||||
FROM harmonization_recipes
|
||||
WHERE recipe_id IN (%s)
|
||||
GROUP BY recipe_id",
|
||||
.sql_lit_chr(candidates)
|
||||
)))
|
||||
}
|
||||
|
||||
#' Attach `ig_recipe_id` to each suggestion: the intergovernmental-expenditure
|
||||
#' recipe (an M-to-local or L-to-state recipe) whose component codes cover
|
||||
#' exactly the same set of function suffixes as the firing recipe's own
|
||||
@@ -272,7 +390,7 @@
|
||||
#' `R/basis.R`). This blocks a recipe surfaced through a mis-scoped
|
||||
#' category from ever reaching the M/L search, e.g. `cog_spending()`'s
|
||||
#' flow_prefixes are `c("E","F","G")`, which `ig_federal_b47_wide`'s own
|
||||
#' `"B"` is not part of.
|
||||
#' "B" is not part of.
|
||||
#' 2. `own_prefix %in% c("E","F","G")`: M/L only ever pairs with the
|
||||
#' DIRECT-expenditure family, never with revenue (`cog_revenue()`'s
|
||||
#' flow_prefixes already fold B/C/D in as ordinary revenue -- there is
|
||||
@@ -280,10 +398,6 @@
|
||||
#' adds one for spending) and never with ANOTHER M/L recipe (without
|
||||
#' this check, `ige_local_m47_wide` would wrongly match sibling
|
||||
#' `ige_state_l47_wide` on their shared {"47","94"} suffix set).
|
||||
#' Condition 1 alone does not catch this: under `cog_revenue()`,
|
||||
#' `ig_federal_b47_wide`'s own `"B"` IS inside revenue's own
|
||||
#' `flow_prefixes`, so only this second, family-specific check blocks
|
||||
#' the search.
|
||||
#' @noRd
|
||||
.attach_ig_counterparts <- function(con, suggestions, flow_prefixes) {
|
||||
if (length(suggestions) == 0L) return(suggestions)
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
# uscogdata
|
||||
|
||||
<!-- badges: start -->
|
||||
[](https://github.com/civilytics/uscogdata/actions/workflows/R-CMD-check.yaml)
|
||||
[](https://civilytics.r-universe.dev/uscogdata)
|
||||
[](LICENSE.md)
|
||||
<!-- badges: end -->
|
||||
@@ -18,7 +19,7 @@ carries provenance describing what was converted, what was aggregated, and
|
||||
which known series breaks intersect your query.
|
||||
|
||||
**Scope:** government types 0–3 (state, county, municipality, township).
|
||||
56 fiscal years, 46,148,034 rows, 190.6 MB. There is no source data for FY1968
|
||||
56 fiscal years, 46,148,034 rows, ~201 MB. There is no source data for FY1968
|
||||
or FY1969. Special districts (type 4) and school districts (type 5) are
|
||||
excluded pending validation.
|
||||
|
||||
@@ -85,14 +86,38 @@ cog_explain(spend)
|
||||
|
||||
| | Remote (default) | Mirrored |
|
||||
|---|---|---|
|
||||
| Setup | none | `cog_mirror(dest)`, 190.6 MB once |
|
||||
| Disk used | **0 MB** — HTTP range requests only | 190.6 MB |
|
||||
| Per query | ~4 s (one government, one year)<br>~6 s (one government, 23 years) | local speed |
|
||||
| Setup | none | `cog_mirror(dest)`, ~201 MB once |
|
||||
| Disk used | **0 MB** — HTTP range requests only | ~201 MB |
|
||||
| Opening a session | ~7.5 s | ~0.1 s |
|
||||
| One government, one year | ~4 s | ~0.05 s |
|
||||
| One government, full history | ~7 s | ~0.1 s |
|
||||
| Later queries, same session | ~1.5 s | ~0.05 s |
|
||||
| Good for | trying it out, teaching, one-off questions | repeated analysis, offline work, reproducibility |
|
||||
|
||||
**A local mirror is roughly 60–80x faster, and it is one function call.** That is
|
||||
by far the largest difference any of these settings makes. If you are going to
|
||||
ask more than a handful of questions, mirror first.
|
||||
|
||||
Measured 2026-08-10 on a 16-core Linux workstation against the published corpus
|
||||
(schema v7, `pipeline_commit 3d28ddd`), fresh R session per arm. A one-off
|
||||
question costs about **12 seconds end to end remotely and 0.15 seconds
|
||||
mirrored**, session setup included.
|
||||
|
||||
Two things the per-query rows hide:
|
||||
|
||||
- **Opening the session is the single largest remote cost** — larger than any
|
||||
one query. It fetches the manifest and registers 23 SQL views over HTTPS, and
|
||||
it lands on your first query, not on `library(uscogdata)`.
|
||||
- **The cost is network round-trips, not scanning.** A repeat query against
|
||||
partitions this session has already touched is ~1.5 s rather than ~4 s, and a
|
||||
full-history query costs ~7 s whether it runs first or last. What you are
|
||||
paying for is reaching each of the 56 yearly files over HTTPS the first time.
|
||||
|
||||
Nothing is written to disk in remote mode: DuckDB fetches the parquet footer,
|
||||
works out which row groups it needs, and reads only those. Nothing is cached
|
||||
between sessions either, so every query goes back to the network.
|
||||
between sessions either, so every query goes back to the network — and a session
|
||||
that issues many remote queries in quick succession can be rate-limited by the
|
||||
host (`HTTP Error: ... 429`). Both are further reasons to mirror for real work.
|
||||
|
||||
The default points at a public HuggingFace mirror of the corpus. If you would
|
||||
rather not depend on a third party — for reproducibility, for an air-gapped
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,14 @@
|
||||
# Project journal
|
||||
|
||||
Append-only, newest first. **Entries are never edited** — the value of this file is
|
||||
that it records what was believed at the time, including the parts that turned out
|
||||
wrong. Where things stand *today* is in `STATUS.md`, which is generated.
|
||||
|
||||
Four lines per entry. The analysis belongs in the issue or the decision record; this
|
||||
file carries the reasoning and the pointers.
|
||||
|
||||
- **Why** — the driver. The one line git cannot reconstruct later.
|
||||
- **Obligates** — issues this change created elsewhere. Numbers, not prose.
|
||||
- **Refs** — commits, issues, decision records.
|
||||
|
||||
---
|
||||
@@ -0,0 +1,67 @@
|
||||
# Project status
|
||||
|
||||
> Between the compass markers is generated. Edit the sources, not this.
|
||||
|
||||
<!-- compass:begin -->
|
||||
<!-- compass:board -->
|
||||
|
||||
## Where this stands
|
||||
|
||||
uscogdata is at 0.4.0 and its public surface is settled: the query verbs, the cohort
|
||||
predicates added in this release, and the provenance contract every verb returns.
|
||||
|
||||
The six open issues split cleanly. Two are API work carried out of the #9 review pass
|
||||
and deliberately deferred there rather than fixed in that branch. Three concern the
|
||||
corpus layer, and the largest of them, partition-level caching, was named the single
|
||||
highest-leverage change on the remote path before being deferred. One, the
|
||||
data-correction intake (#52), is a decision rather than a task: it was parked during
|
||||
the 0.3.0 design, and the API announcement waits on it, because without it the corpus
|
||||
cannot make the "traceable and correctable" claim that most distinguishes it from
|
||||
Census's own files.
|
||||
|
||||
Nothing here is blocked on anything else, so the ordering is a judgement about value
|
||||
rather than a dependency graph.
|
||||
|
||||
Compass's own files moved out of `docs/` this session. They were sitting inside
|
||||
pkgdown's output directory, and `pkgdown::clean_site()` deletes every top-level entry
|
||||
there except `CNAME` and `dev` — asked directly, it listed `docs/pm` and
|
||||
`docs/decisions` among the 28 it would remove, with the guard that would have stopped
|
||||
it satisfied by `docs/pkgdown.yml`. They are in `pm/` now. Nothing was lost: the
|
||||
journal had no entries and there were no decision records yet, which made this the
|
||||
cheapest moment to move. The `.gitignore` workaround that re-included two children of
|
||||
an excluded `docs/` is gone with it.
|
||||
|
||||
## Ready to work on next
|
||||
|
||||
- **#34** cog_revenue() offers expenditure recipes as suggestions: scope the candidate query by category_type · `ws/api` — nothing is blocking it; something is currently wrong
|
||||
- **#36** n_units_reporting is category-conditional and cannot be read as a response rate · `ws/corpus` — nothing is blocking it; owed work from an earlier change
|
||||
- **#2** Extend population data to be households as an alternate spending denominator · `ws/corpus` — nothing is blocking it
|
||||
- **#33** Decompose .build_suggestions() (106 lines) into named helpers · `ws/api` — nothing is blocking it
|
||||
- **#52** Release 11/11: design the data-correction intake (deferred; gates the API announcement) · `ws/corpus` — nothing is blocking it
|
||||
- **#64** Partition-level caching: R/cache.R is still a stub, and the remote path pays for it every session · `ws/corpus` — nothing is blocking it
|
||||
|
||||
## Workstreams
|
||||
|
||||
| Stream | Commits since | Open | Debt | Owes docs |
|
||||
|---|---|---|---|---|
|
||||
| Query verbs and results | 77 | 2 | 0 | no |
|
||||
| Corpus, mirror, provenance | 39 | 4 | 1 | no |
|
||||
| Vignettes and guides | 34 | 0 | 0 | **yes** |
|
||||
|
||||
## CI
|
||||
|
||||

|
||||

|
||||
|
||||
<details>
|
||||
<summary>Dependency graph and detail</summary>
|
||||
|
||||
_Nothing blocks anything else, so there is no graph to draw._
|
||||
|
||||
- Marker: `none` (no journal entry yet)
|
||||
- Commits since: 165
|
||||
- Open issues: 6
|
||||
|
||||
</details>
|
||||
|
||||
<!-- compass:end -->
|
||||
@@ -0,0 +1,44 @@
|
||||
[project]
|
||||
name = "uscogdata"
|
||||
forge = "Civilytics/uscogdata"
|
||||
|
||||
# Three strands that go stale independently: what the verbs return, what the
|
||||
# corpus is and how it is mounted, and how both are explained to a reader.
|
||||
|
||||
[[workstream]]
|
||||
id = "api"
|
||||
title = "Query verbs and results"
|
||||
paths = [
|
||||
"R/revenue.R", "R/spending.R", "R/balances.R", "R/peers.R", "R/search.R",
|
||||
"R/categories.R", "R/recipes.R", "R/rollup.R", "R/explain.R", "R/basket.R",
|
||||
"R/suggestions.R", "R/suppression.R", "R/complete.R", "R/cohort.R",
|
||||
"R/basis.R", "R/adjust.R", "R/pagination.R",
|
||||
]
|
||||
docs = ["vignettes/*.Rmd", "README.md"]
|
||||
|
||||
[[workstream]]
|
||||
id = "corpus"
|
||||
title = "Corpus, mirror, provenance"
|
||||
paths = [
|
||||
"R/manifest.R", "R/mirror.R", "R/cache.R", "R/session.R", "R/provenance.R",
|
||||
"R/coverage.R", "R/config.R", "R/views.R", "R/series_breaks.R",
|
||||
"R/balance_caveats.R", "R/zzz.R", "data-raw/**", "inst/sql/**",
|
||||
]
|
||||
docs = ["vignettes/*.Rmd", "NEWS.md"]
|
||||
|
||||
[[workstream]]
|
||||
id = "docs"
|
||||
title = "Vignettes and guides"
|
||||
paths = ["vignettes/**", "README.md", "_pkgdown.yml", "NEWS.md"]
|
||||
docs = []
|
||||
|
||||
[roborev]
|
||||
project_guidelines = [
|
||||
"Every verb calls .ensure_session() first, then queries via DBI::dbGetQuery().",
|
||||
"A verb's return value is always a tbl_df carrying a provenance attribute.",
|
||||
"govid inputs always go through .coerce_govid_input(); it accepts a character vector or a data frame.",
|
||||
"SQL has two layers: view definitions are numbered .sql files in inst/sql/ registered by .register_views(); query construction is inline sprintf() in R. Add a view as a file; build a query in R.",
|
||||
"No arrow dependency -- DuckDB reads parquet natively.",
|
||||
"withr is Suggests-only and must appear in tests alone.",
|
||||
"Tests must pass offline against the bundled fixture; tests/testthat/setup.R sets USCOGDATA_URL for that.",
|
||||
]
|
||||
@@ -0,0 +1,12 @@
|
||||
# Decisions
|
||||
|
||||
One file per decision, numbered and immutable. A decision that changes is superseded
|
||||
by a new record, never edited in place — the old reasoning is the point.
|
||||
|
||||
The table below is **generated** by `compass:decide`. Do not hand-edit it.
|
||||
|
||||
<!-- compass:begin decisions -->
|
||||
| # | Date | Decision | Status |
|
||||
|---|---|---|---|
|
||||
| — | — | *No decisions recorded yet.* | — |
|
||||
<!-- compass:end decisions -->
|
||||
@@ -315,15 +315,17 @@ test_that("a mis-scoped cog_spending() call never attaches an M/L counterpart to
|
||||
# (SB194, cog_pipeline#64), so the recipe stopped being a candidate there.
|
||||
# FL state carries a real FY2011 B47 amount, so this exercises the guard
|
||||
# against a suggestion that genuinely fires.
|
||||
#
|
||||
# Issue #34: "IG Federal" maps to B-prefixed codes in summary_categories
|
||||
# with category_type = 'revenue'. A spending verb (flow_prefixes E/F/G)
|
||||
# now scopes its candidate query by category_type = 'expenditure', so it
|
||||
# correctly finds NO candidates for this revenue-only category -- the
|
||||
# suggestion machinery cannot fire, and no M/L counterpart is attached.
|
||||
r <- suppressMessages(
|
||||
cog_spending("120000226351", years = c(2005, 2011), category = "IG Federal")
|
||||
)
|
||||
sugg <- attr(r, "provenance")$suggestions
|
||||
expect_gt(length(sugg), 0L)
|
||||
ids <- vapply(sugg, function(s) s$recipe_id %||% "", character(1))
|
||||
expect_true("ig_federal_b47_wide" %in% ids)
|
||||
ig <- unlist(lapply(sugg, function(s) s$ig_recipe_id))
|
||||
expect_length(ig, 0L)
|
||||
expect_length(sugg, 0L)
|
||||
})
|
||||
|
||||
test_that("C1: 'total' on a legacy aggregate-only family reports the IG-only figure honestly, not as Direct + IG", {
|
||||
|
||||
@@ -371,34 +371,45 @@ test_that("uscogdata#9: the revenue verb inherits the same trigger", {
|
||||
expect_null(sugg[[1]]$ig_recipe_id)
|
||||
})
|
||||
|
||||
test_that("I1: cog_revenue never fabricates suppressed dollars for an expenditure-only recipe", {
|
||||
test_that("I1 + #34: cog_revenue never suggests expenditure-only recipes", {
|
||||
# uscogdata#9 review, finding I1: Corrections is an expenditure-only
|
||||
# category (E04/E05). cog_revenue() naturally returns zero rows for it, so
|
||||
# corrections_combined still fires as an empty_year suggestion (its own
|
||||
# generic join finds real E04/E05 data for this government) -- but before
|
||||
# the flow_prefixes fix, .suppressed_components() measured E04/E05 against
|
||||
# cog_revenue()'s OWN view (which can never contain an E-coded row by
|
||||
# construction) and reported the full $3,631,945,000 as "suppressed",
|
||||
# when cog_spending() for the same gov/years/category actually returns
|
||||
# $3,691,029,000 -- nothing was suppressed at all.
|
||||
# category (E04/E05). Before the flow_prefixes fix (#9), .suppressed_components()
|
||||
# measured E04/E05 against cog_revenue()'s OWN view and reported $3.6B as
|
||||
# "suppressed" -- nothing was suppressed at all.
|
||||
#
|
||||
# Issue #34 builds on that: the candidate query now also filters by
|
||||
# category_type ('revenue'), so expenditure-only recipes like corrections_combined
|
||||
# (whose components E04/E05 are classified as 'expenditure' in summary_categories)
|
||||
# are never even considered for a revenue verb. This is stronger than just
|
||||
# suppressing the dollar claim -- it prevents the suggestion from firing at all.
|
||||
skip_if_no_corpus()
|
||||
r <- suppressMessages(
|
||||
cog_revenue("061037123085", years = 2019:2020, category = "Corrections"))
|
||||
sugg <- attr(r, "provenance")$suggestions
|
||||
ids <- vapply(sugg, function(s) s$recipe_id, character(1))
|
||||
expect_true("corrections_combined" %in% ids)
|
||||
ids <- vapply(sugg, function(s) s$recipe_id %||% "", character(1))
|
||||
|
||||
hit <- sugg[[which(ids == "corrections_combined")]]
|
||||
expect_equal(hit$suppressed_amount, 0)
|
||||
expect_equal(hit$suppressed_years, integer(0))
|
||||
expect_equal(hit$suppressed_codes, character(0))
|
||||
# corrections_combined should NOT appear -- its components are expenditure-only.
|
||||
expect_false("corrections_combined" %in% ids)
|
||||
})
|
||||
|
||||
# And cog_spending() for the identical gov/years/category is unaffected --
|
||||
# it actually finds the E04/E05 dollars the buggy measurement claimed were
|
||||
# excluded.
|
||||
sp <- suppressMessages(
|
||||
cog_spending("061037123085", years = 2019:2020, category = "Corrections"))
|
||||
expect_equal(sum(sp$amt_nominal), 3691029000)
|
||||
test_that(".query_candidate_recipes() scopes candidates by category_type (#34)", {
|
||||
# Direct assertion on the mechanism the two tests above exercise
|
||||
# end-to-end: corrections_combined's own components (E04/E05) are
|
||||
# category_type = 'expenditure' in summary_categories, so an
|
||||
# expenditure-flavored flow_prefixes call must surface it and a
|
||||
# revenue-flavored one must not. This queries only summary_categories/
|
||||
# harmonization_recipes (no government data), so it runs against the
|
||||
# bundled fixture with no skip_if_no_corpus() needed.
|
||||
con <- uscogdata:::.ensure_session()
|
||||
|
||||
expenditure <- uscogdata:::.query_candidate_recipes(
|
||||
con, category = "Corrections", flow_prefixes = c("E", "F", "G"))
|
||||
expect_true("corrections_combined" %in% expenditure)
|
||||
|
||||
revenue <- uscogdata:::.query_candidate_recipes(
|
||||
con, category = "Corrections",
|
||||
flow_prefixes = c("T", "A", "U", "B", "C", "D"))
|
||||
expect_false("corrections_combined" %in% revenue)
|
||||
})
|
||||
|
||||
test_that("uscogdata#9: no partial-coverage fire in a modern year", {
|
||||
|
||||
Reference in New Issue
Block a user