Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4bedf857e9
|
||
|
|
a9de5ba5ab
|
||
|
|
d0b4bae3cc
|
+3
-4
@@ -3,18 +3,17 @@
|
||||
^\.Rproj\.user$
|
||||
^_pkgdown\.yml$
|
||||
^docs$
|
||||
^pm$
|
||||
^Meta$
|
||||
^doc$
|
||||
^pkgdown$
|
||||
^\.github$
|
||||
^LICENSE\.md$
|
||||
^\.git$
|
||||
^\.gitignore$
|
||||
\.gitkeep$
|
||||
^vignettes$
|
||||
^specs$
|
||||
^plans$
|
||||
^doc$
|
||||
^Meta$
|
||||
^\.gitea$
|
||||
^CLAUDE\.md$
|
||||
^\.superpowers$
|
||||
^CONTRIBUTING\.md$
|
||||
|
||||
@@ -1,55 +0,0 @@
|
||||
# Mirror the canonical Gitea repo to the public GitHub mirror.
|
||||
#
|
||||
# Deliberately a plain `git push`, NOT Gitea's built-in push mirror. A push
|
||||
# mirror force-updates the refs it owns: if anyone ever clicks Merge on a
|
||||
# GitHub PR, the next sync silently overwrites main, the PR still displays
|
||||
# "Merged", the commit becomes unreachable, and nothing anywhere says so.
|
||||
# A non-force push is REJECTED as non-fast-forward the moment that happens,
|
||||
# turning a silent data-loss trap into a red CI run in a place we already look.
|
||||
#
|
||||
# Do NOT add --force here, and do NOT add GitHub branch protection to the
|
||||
# mirror: protection rules block the mirror's legitimate pushes too, breaking
|
||||
# normal syncing to catch an abnormal case.
|
||||
#
|
||||
# PAT_GH is a GitHub personal access token (repo + workflow scope; workflow is
|
||||
# required because this pushes .github/workflows/). It is stored as a Gitea
|
||||
# Actions secret. The name cannot begin with GITHUB_ or GITEA_ -- Gitea
|
||||
# reserves both prefixes for its own injected variables and rejects the secret.
|
||||
name: Mirror to GitHub
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
tags: ['v*']
|
||||
|
||||
jobs:
|
||||
mirror:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Push main and tags to the GitHub mirror
|
||||
env:
|
||||
PAT_GH: ${{ secrets.PAT_GH }}
|
||||
run: |
|
||||
set -eu
|
||||
if [ -z "${PAT_GH:-}" ]; then
|
||||
echo "PAT_GH is unset -- add it under Settings > Actions > Secrets." >&2
|
||||
exit 1
|
||||
fi
|
||||
# On a tag-triggered run, checkout materializes refs/tags/<tag> as a
|
||||
# LIGHTWEIGHT tag at the commit SHA -- the annotated tag object Gitea
|
||||
# holds is never fetched. Mirroring that strips the annotation, and the
|
||||
# NEXT run on main (which does fetch the real object) is then rejected
|
||||
# with "already exists" trying to correct it, because git will not
|
||||
# clobber an existing tag. That is why v0.4.0 failed to mirror.
|
||||
#
|
||||
# Re-fetch canonical tag objects from Gitea first. --force here rewrites
|
||||
# LOCAL tag refs only; it is not a force push and does not weaken the
|
||||
# non-force guarantee on main documented above.
|
||||
git fetch --tags --force origin
|
||||
|
||||
git push "https://x-access-token:${PAT_GH}@github.com/civilytics/uscogdata.git" \
|
||||
HEAD:refs/heads/main --tags
|
||||
@@ -1,61 +0,0 @@
|
||||
# Multi-platform R CMD check, running on the GitHub mirror.
|
||||
#
|
||||
# This exists because the canonical Gitea runner is Linux-only, and this
|
||||
# package hard-depends on duckdb and httr2 -- both compiled, both with real
|
||||
# platform variance -- while having never been checked on Windows or macOS.
|
||||
# A large share of the audience is on Windows.
|
||||
#
|
||||
# Gitea reads .gitea/workflows and GitHub reads .github/workflows, so this
|
||||
# file is inert on the canonical repo and coexists with the Gitea CI that
|
||||
# remains authoritative for deploys.
|
||||
#
|
||||
# The suite needs NO credentials: tests/testthat/setup.R points USCOGDATA_URL
|
||||
# at the bundled fixture corpus. That is exactly why inst/extdata/fixture_corpus
|
||||
# must never be added to .Rbuildignore.
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
pull_request:
|
||||
|
||||
name: R-CMD-check
|
||||
|
||||
permissions: read-all
|
||||
|
||||
jobs:
|
||||
R-CMD-check:
|
||||
runs-on: ${{ matrix.config.os }}
|
||||
name: ${{ matrix.config.os }} (${{ matrix.config.r }})
|
||||
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
config:
|
||||
- {os: macos-latest, r: 'release'}
|
||||
- {os: windows-latest, r: 'release'}
|
||||
- {os: ubuntu-latest, r: 'devel', http-user-agent: 'release'}
|
||||
- {os: ubuntu-latest, r: 'release'}
|
||||
|
||||
env:
|
||||
GITHUB_PAT: ${{ secrets.GITHUB_TOKEN }}
|
||||
R_KEEP_PKG_SOURCE: yes
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: r-lib/actions/setup-pandoc@v2
|
||||
|
||||
- uses: r-lib/actions/setup-r@v2
|
||||
with:
|
||||
r-version: ${{ matrix.config.r }}
|
||||
http-user-agent: ${{ matrix.config.http-user-agent }}
|
||||
use-public-rspm: true
|
||||
|
||||
- uses: r-lib/actions/setup-r-dependencies@v2
|
||||
with:
|
||||
extra-packages: any::rcmdcheck
|
||||
needs: check
|
||||
|
||||
- uses: r-lib/actions/check-r-package@v2
|
||||
with:
|
||||
upload-snapshots: true
|
||||
build_args: 'c("--no-manual")'
|
||||
@@ -1,64 +0,0 @@
|
||||
# Explain the mirror contribution flow on every incoming pull request.
|
||||
#
|
||||
# This repository is a MIRROR. A PR opened here is landed on the canonical Gitea
|
||||
# repository and syncs back; because the merge preserves the contributor's
|
||||
# commits at their original SHAs, GitHub marks the PR "Merged" on its own as
|
||||
# soon as the mirror syncs -- with nobody visibly clicking Merge.
|
||||
#
|
||||
# Without this comment, that reads as a rejection: the contributor sees their PR
|
||||
# close with no review, no merge button pressed, and no explanation. It is
|
||||
# actually the successful outcome. Say so up front, before it happens.
|
||||
#
|
||||
# WHY pull_request_target AND NOT pull_request:
|
||||
# a `pull_request` run from a fork gets a read-only token, so it cannot post a
|
||||
# comment -- which is exactly the case this workflow exists to serve.
|
||||
# `pull_request_target` runs in the context of the BASE repo and gets a writable
|
||||
# token. That is only safe because this job never checks out or executes the
|
||||
# contributor's code; it posts a fixed string. Do not add a checkout of
|
||||
# `github.event.pull_request.head.sha` here -- that combination is the standard
|
||||
# pull_request_target privilege-escalation hole.
|
||||
name: Explain the mirror flow
|
||||
|
||||
on:
|
||||
pull_request_target:
|
||||
types: [opened]
|
||||
|
||||
permissions:
|
||||
pull-requests: write
|
||||
|
||||
jobs:
|
||||
comment:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Post the contribution-flow explainer
|
||||
uses: actions/github-script@v7
|
||||
with:
|
||||
script: |
|
||||
const body = [
|
||||
"Thanks for this — and one thing worth knowing before it happens.",
|
||||
"",
|
||||
"**This repository is a mirror.** Development happens on Gitea at",
|
||||
"`gitea.civilytics.org/Civilytics/uscogdata`. Your pull request will be fetched",
|
||||
"from here, landed there, and synced back.",
|
||||
"",
|
||||
"Because that merge preserves your commits at their original SHAs, **GitHub will",
|
||||
"mark this pull request \"Merged\" on its own** — without anyone visibly clicking",
|
||||
"the Merge button, and possibly without a review comment on this page first.",
|
||||
"",
|
||||
"> If your pull request closes as \"Merged\" and nobody appears to have merged it,",
|
||||
"> that is the normal, successful outcome — not a rejection.",
|
||||
"",
|
||||
"If it is *not* going to be merged, you will get an actual reply saying so.",
|
||||
"",
|
||||
"Substantial contributions get a `ctb` entry in `DESCRIPTION`, which surfaces in",
|
||||
"`citation(\"uscogdata\")`. There is no CLA and no DCO sign-off.",
|
||||
"",
|
||||
"Full details: [CONTRIBUTING.md](https://github.com/civilytics/uscogdata/blob/main/CONTRIBUTING.md).",
|
||||
].join("\n");
|
||||
|
||||
await github.rest.issues.createComment({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
issue_number: context.payload.pull_request.number,
|
||||
body,
|
||||
});
|
||||
@@ -4,10 +4,6 @@
|
||||
.Ruserdata
|
||||
*.Rproj
|
||||
inst/doc
|
||||
# pkgdown output. Compass used to keep its files in docs/pm/ and
|
||||
# docs/decisions/, which forced this to be written as children with two
|
||||
# re-includes -- git cannot re-include anything beneath an excluded directory.
|
||||
# Compass lives in pm/ now, so the whole directory can be excluded again.
|
||||
docs/
|
||||
/doc/
|
||||
/Meta/
|
||||
@@ -16,6 +12,3 @@ docs/
|
||||
|
||||
# SDD working artifacts (ledger, briefs, review packages) — plans/ stays tracked
|
||||
.superpowers/sdd/
|
||||
.compass-cache/
|
||||
# roborev snapshots
|
||||
/.roborev/
|
||||
|
||||
@@ -1,60 +0,0 @@
|
||||
# roborev configuration, initialised by compass.
|
||||
# Reviews are queued to a background daemon -- they never block a commit.
|
||||
|
||||
post_commit_review = 'commit'
|
||||
excluded_commit_patterns = ['WIP', 'chore:', 'chore(', 'docs:', 'Merge ']
|
||||
|
||||
review_guidelines = '''
|
||||
# --- compass:begin (generated -- edit the sources, not this) ---
|
||||
- Prefer returning new values to mutating arguments in place. A function that edits
|
||||
its caller's object is a bug waiting for a second caller.
|
||||
- Validate at system boundaries -- user input, API responses, file contents, config.
|
||||
Fail fast with a message naming the field and the file.
|
||||
- Never swallow an error. Handle it or let it propagate; a bare catch that continues
|
||||
is worse than a crash.
|
||||
- No hardcoded secrets, tokens, or credentials, and no secrets in log output or error
|
||||
messages.
|
||||
- Parameterise every query. String-built SQL is a defect even when the input looks safe.
|
||||
- Keep functions under roughly 50 lines and files under roughly 400. Flag nesting
|
||||
deeper than four levels.
|
||||
- No magic numbers or hardcoded paths -- name them as constants or read them from config.
|
||||
- New behaviour needs a test. A bug fix needs a test that fails without the fix.
|
||||
- Prose a person reads -- an issue title or body, a journal entry, a decision record,
|
||||
the narrative on the status board -- names the action or the thing, not the shape of
|
||||
the machinery. Flag "gate", "seam", "surface area", "load-bearing", "first-class",
|
||||
"primitive", "blast radius". A project's own defined vocabulary is not the target.
|
||||
- Use the native pipe `|>`, not magrittr `%>%`.
|
||||
- snake_case for objects and functions; UPPER_SNAKE for constants. Never use `.` as a
|
||||
word separator in a function name -- it collides with S3 dispatch.
|
||||
- Validate arguments at the top of exported functions with `stopifnot()` or an explicit
|
||||
check, and say which argument was wrong.
|
||||
- Never `setDT()`, `set()`, or otherwise modify by reference a data.table the caller
|
||||
still owns. `as.data.table()` copies; use it.
|
||||
- Prefer `vapply()` to `sapply()` -- `sapply()` silently returns a list when the type
|
||||
varies, which turns a type error into a downstream mystery.
|
||||
- Use `seq_len(n)` / `seq_along(x)`, never `1:n`, which iterates backwards when n is 0.
|
||||
- Compare strings with `==` only after checking for NA; use `identical()` for scalars
|
||||
where NA would be wrong.
|
||||
- Do not call `library()` inside package or module files; attach packages in scripts and
|
||||
test helpers only.
|
||||
- Namespace-qualify calls into other packages (`stats::sd`) in code that is sourced.
|
||||
- Every exported function needs roxygen with `@param` for each argument (type, meaning,
|
||||
and why the default is what it is) and `@return`. Add `@examples` for exported API.
|
||||
- Declare dependencies in DESCRIPTION. Prefer base R or an existing dependency over
|
||||
adding a new one; a package with zero hard deps is worth keeping that way.
|
||||
- Signal errors with `stop()` carrying a condition class, so callers can catch the kind
|
||||
rather than matching on message text.
|
||||
- Keep internals internal. Export only what a user needs; an accidentally exported
|
||||
helper becomes an API you have to keep.
|
||||
- Tests use testthat edition 3. Each test is self-sufficient -- no reliance on state
|
||||
left by an earlier test or on a fixture built elsewhere in the file.
|
||||
- Prefer duplication in tests over a helper that hides what is being asserted.
|
||||
- Every verb calls .ensure_session() first, then queries via DBI::dbGetQuery().
|
||||
- A verb's return value is always a tbl_df carrying a provenance attribute.
|
||||
- govid inputs always go through .coerce_govid_input(); it accepts a character vector or a data frame.
|
||||
- SQL has two layers: view definitions are numbered .sql files in inst/sql/ registered by .register_views(); query construction is inline sprintf() in R. Add a view as a file; build a query in R.
|
||||
- No arrow dependency -- DuckDB reads parquet natively.
|
||||
- withr is Suggests-only and must appear in tests alone.
|
||||
- Tests must pass offline against the bundled fixture; tests/testthat/setup.R sets USCOGDATA_URL for that.
|
||||
# --- compass:end ---
|
||||
'''
|
||||
@@ -124,10 +124,3 @@ devtools::test()
|
||||
a numbered `.sql` file; build a query in R.
|
||||
- No arrow dependency — DuckDB reads parquet natively
|
||||
- `withr` is a Suggests-only dep; only used in tests
|
||||
|
||||
## Domain context — read this first
|
||||
|
||||
**Before doing any work in this repo, read `~/.claude/memory/values/civilytics.md`.**
|
||||
It carries the purpose, direction, and constraints for this domain. It is not optional
|
||||
context — read it before planning or writing code, not after. (An `@` import will not
|
||||
work here; project-level imports don't preload. The read is the mechanism.)
|
||||
|
||||
-113
@@ -1,113 +0,0 @@
|
||||
# Contributing to uscogdata
|
||||
|
||||
Thanks for reading this — a package like this gets better mostly through people
|
||||
noticing that a number looks wrong.
|
||||
|
||||
## Where the code lives
|
||||
|
||||
Development happens on **Gitea**, at
|
||||
`gitea.civilytics.org/Civilytics/uscogdata`. The repository at
|
||||
`github.com/civilytics/uscogdata` is a **mirror** that accepts issues and pull
|
||||
requests.
|
||||
|
||||
## What happens to a GitHub pull request
|
||||
|
||||
Open it normally. Behind the scenes it is fetched and landed on the canonical
|
||||
Gitea repository, then syncs back:
|
||||
|
||||
```sh
|
||||
git fetch github refs/pull/42/head:pr-42
|
||||
git switch main && git merge --no-ff pr-42
|
||||
git push origin main # Gitea -> mirror -> GitHub
|
||||
```
|
||||
|
||||
Because the merge preserves your commits at their original SHAs, **GitHub marks
|
||||
your PR merged on its own** as soon as the mirror syncs. So:
|
||||
|
||||
> If your pull request closes as "Merged" without anyone visibly clicking
|
||||
> Merge, that is the normal, successful outcome — not a rejection.
|
||||
|
||||
Substantial contributions get a `ctb` entry in `DESCRIPTION`, which surfaces in
|
||||
`citation("uscogdata")`.
|
||||
|
||||
There is no CLA and no DCO sign-off requirement.
|
||||
|
||||
## Running the tests
|
||||
|
||||
```r
|
||||
devtools::test() # bundled fixture; no network, no credentials
|
||||
```
|
||||
|
||||
`tests/testthat/setup.R` points `USCOGDATA_URL` at
|
||||
`inst/extdata/fixture_corpus/` automatically — a four-year slice (2011, 2012,
|
||||
2019, 2020) covering all 50 states. That is the whole data setup.
|
||||
|
||||
## Testing against the live corpus
|
||||
|
||||
```sh
|
||||
USCOGDATA_LIVE_TEST=true Rscript -e 'devtools::test(filter = "live-corpus")'
|
||||
```
|
||||
|
||||
This is worth understanding rather than skipping. Until 0.3.0 the package
|
||||
**could not read a remote corpus at all** — the partitioned view used a glob,
|
||||
and DuckDB cannot expand a glob over generic HTTP. It went unnoticed for months
|
||||
because every test path used a local corpus (the bundled fixture), and so did
|
||||
the production API (a host mount). Nothing exercised the package the way a new
|
||||
user does.
|
||||
|
||||
`test-live-corpus.R` is the only test that runs with no `USCOGDATA_URL`, no
|
||||
option, and no fixture. If you change anything touching view registration,
|
||||
manifest handling, or configuration, run it.
|
||||
|
||||
## Do not exclude the fixture from the build
|
||||
|
||||
There is a temptation to add `^inst/extdata/fixture_corpus$` to
|
||||
`.Rbuildignore` because 15 MB feels large for a package. Don't:
|
||||
|
||||
- `vignette("total-spending")` reads from it and would fail to build.
|
||||
- `R CMD check` on r-universe and GitHub Actions would have no corpus, so the
|
||||
suite could not run without credentials.
|
||||
|
||||
This package is not going to CRAN, so its 5 MB guidance does not apply. A
|
||||
package-size NOTE in `R CMD check` is expected and acceptable.
|
||||
|
||||
## Downstream consumers
|
||||
|
||||
`cog-api` depends on this package and its CI clones uscogdata at
|
||||
`USCOGDATA_REF`, **defaulting to `main`**. There is no pin. Anything merged
|
||||
here reaches the API's next build, so before merging a change to the reader,
|
||||
run the API suite against your branch:
|
||||
|
||||
```sh
|
||||
Rscript -e "remotes::install_local('/path/to/uscogdata', upgrade = 'never')"
|
||||
cd /path/to/cog-api/api/tests/testthat
|
||||
Rscript -e 'testthat::test_dir(".", stop_on_failure = TRUE)'
|
||||
```
|
||||
|
||||
The API calls only exported verbs, so internal refactors are usually safe —
|
||||
but "usually" is not a release gate.
|
||||
|
||||
## Release checklist
|
||||
|
||||
1. `devtools::test()` — green against the bundled fixture, offline.
|
||||
2. `USCOGDATA_LIVE_TEST=true devtools::test()` — green against the live corpus.
|
||||
3. cog-api suite green against this branch (above).
|
||||
4. `devtools::check(args = "--as-cran")` — 0 errors, 0 warnings.
|
||||
5. `pkgdown::build_site()` completes.
|
||||
6. Vignettes resolve from an installed copy:
|
||||
`vignette("total-spending", package = "uscogdata")`.
|
||||
7. **Cold-start check**: on a machine that has never had this package,
|
||||
install it and run the README quickstart verbatim with no environment
|
||||
variables set. This is the only check that catches a
|
||||
corpus-unreachable defect, and its absence is why 0.3.0 needed fixing.
|
||||
8. Bump `Version` and add a `NEWS.md` section.
|
||||
9. Tag on **Gitea** (`git tag -a vX.Y.Z && git push origin vX.Y.Z`). The mirror
|
||||
workflow carries tags to GitHub on its own — confirm the tag appears at
|
||||
`github.com/civilytics/uscogdata/tags` before continuing.
|
||||
10. Update the r-universe registry pin at
|
||||
`github.com/civilytics/civilytics.r-universe.dev` — edit `packages.json`'s
|
||||
`branch` to the new tag. **r-universe will not pick up a release until this
|
||||
is edited**: the pin is a tag, deliberately, so a mid-refactor `main` is
|
||||
never published as a release. `"branch": "*release"` would track releases
|
||||
automatically, but it needs a GitHub *Release* object and the mirror pushes
|
||||
tags only — so it would silently never update.
|
||||
+4
-11
@@ -1,21 +1,14 @@
|
||||
Package: uscogdata
|
||||
Type: Package
|
||||
Title: Curated Reader for the Civilytics US Census of Governments Finance Corpus
|
||||
Version: 0.4.0
|
||||
Authors@R: c(
|
||||
person(c("Jared", "E."), "Knowles",
|
||||
email = "jared@civilytics.com",
|
||||
role = c("aut", "cre"),
|
||||
comment = c(ORCID = "0000-0003-0005-9478")),
|
||||
person("Civilytics Consulting LLC", role = c("cph", "fnd")))
|
||||
Version: 0.1.0
|
||||
Authors@R:
|
||||
person("Civilytics", , , "jknowles@gmail.com", role = c("aut", "cre"))
|
||||
Description: Curated R verbs over the Civilytics US Census of Governments
|
||||
finance corpus. Provides unit-level financial profiles, geographic
|
||||
rollups, and peer comparisons with auditable provenance and built-in
|
||||
cross-vintage correctness.
|
||||
License: MIT + file LICENSE
|
||||
URL: https://github.com/civilytics/uscogdata,
|
||||
https://civilytics.r-universe.dev/uscogdata
|
||||
BugReports: https://github.com/civilytics/uscogdata/issues
|
||||
Encoding: UTF-8
|
||||
LazyData: false
|
||||
Depends: R (>= 4.1)
|
||||
@@ -39,4 +32,4 @@ Config/testthat/edition: 3
|
||||
VignetteBuilder: knitr
|
||||
RoxygenNote: 7.3.3
|
||||
MinCorpusSchema: 4
|
||||
MaxCorpusSchema: 7
|
||||
MaxCorpusSchema: 5
|
||||
|
||||
@@ -1,2 +1,2 @@
|
||||
YEAR: 2026
|
||||
COPYRIGHT HOLDER: Civilytics Consulting LLC
|
||||
COPYRIGHT HOLDER: Civilytics
|
||||
|
||||
-21
@@ -1,21 +0,0 @@
|
||||
# MIT License
|
||||
|
||||
Copyright (c) 2026 Civilytics Consulting LLC
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
@@ -1,230 +1,225 @@
|
||||
# uscogdata 0.4.0
|
||||
# uscogdata 0.1.0 (development)
|
||||
|
||||
## Cohorts can be named by predicate, not just by id
|
||||
## New: `cog_balances()` for cash-and-security holdings
|
||||
|
||||
`cog_spending()`, `cog_revenue()` and `cog_balances()` gain optional `state`
|
||||
and `type` arguments. Both default to `NULL`, so every existing call behaves
|
||||
exactly as before.
|
||||
* New `cog_balances()` exposes the 14 cash-and-security holding codes
|
||||
(`category_type = "balance"`): fund balances, retirement system holdings and
|
||||
insurance trust balances (#25). Holdings are a stock, not a flow, so the verb
|
||||
has no `expenditure_concept` / `revenue_concept` / `complete` arguments, and
|
||||
no `subtype` argument either -- for holdings, `category` is a strict
|
||||
coarsening of `balance_subtype`, so `category = "Fund Balances"` is exactly
|
||||
the `general` family (`W01`/`W31`/`W61`).
|
||||
* `cog_balances()` results carry `provenance$balance_caveats`, recording that
|
||||
Census holdings are gross rather than GAAP fund balance, and the measured
|
||||
coverage window of each subtype family.
|
||||
|
||||
Passing them expresses the cohort as a subquery against `canonical_fips_xwalk`
|
||||
inside each statement, instead of round-tripping the ids through R and
|
||||
rendering them back into a literal `IN` list:
|
||||
## Multi-government aggregates now disclose their reporting coverage
|
||||
|
||||
```r
|
||||
# before: resolve 20,106 ids in R, then embed them in every statement
|
||||
ids <- cog_gov_search(NULL, state = "CA", type = "city")$canonical_govid
|
||||
cog_spending(ids, years = 2022)
|
||||
* The Census of Governments is a **complete census only in years ending in 2
|
||||
and 7**; every other year is a sample, and the sample varies enormously. On
|
||||
the bundled fixture, Wisconsin's 608-city universe rolls up **597**
|
||||
governments in FY2012 and **112** in FY2019 — an 18%-to-98% swing the
|
||||
return value said nothing about, so a statewide total resting on a fifth of
|
||||
the universe looked exactly like one resting on all of it.
|
||||
* `cog_geographic_rollup()`, `cog_peer_compare()` and `cog_find_peers()` gain
|
||||
`coverage`:
|
||||
|
||||
# now: the cohort never leaves the database
|
||||
cog_spending(years = 2022, state = "CA", type = "city")
|
||||
```
|
||||
| value | effect |
|
||||
|---|---|
|
||||
| `"all"` (default) | every unit that reported that year — unchanged behaviour |
|
||||
| `"census"` | census years only; aborts if the range holds none rather than returning nothing |
|
||||
| `"consistent"` | only units reporting in *every* requested year — a balanced panel |
|
||||
|
||||
Measured against the production corpus, same FY2022 aggregate over the
|
||||
20,106-government `type = "city"` cohort:
|
||||
* **Regardless of mode**, every result now carries `provenance$coverage` with
|
||||
per-year `n_units_reporting`, `n_units_expected` and `is_census_year`, plus
|
||||
`provenance$coverage_mode`. `cog_explain()` prints a "Reporting coverage"
|
||||
section. So the default mode can no longer mislead silently.
|
||||
* `is_census_year` is a statement about the **survey calendar**, never a claim
|
||||
of completeness: FY1967 is a census year in which only 97 of Wisconsin's 608
|
||||
cities report. `n_units_reporting` is the number that tells the truth.
|
||||
* On `cog_peer_compare()` the target is exempt from `"consistent"` balancing —
|
||||
it is the subject of the comparison, not a member of the cohort — and the
|
||||
`summary_*` quantiles are computed after the filter, so they describe the
|
||||
cohort actually returned. `n_units_reporting` counts peers only, against the
|
||||
cohort size.
|
||||
* On `cog_find_peers()`, `coverage` governs the cohort **vintage** when `year`
|
||||
is `NULL`: `"census"` snaps to the most recent census year with an observed
|
||||
population, so a cohort is not built from a sample year in which most of the
|
||||
candidate universe is absent.
|
||||
|
||||
| cohort expressed as | time |
|
||||
|---|---:|
|
||||
| `IN (20,106 literals)` | 449 ms |
|
||||
| join against a temp cohort table | 99 ms |
|
||||
| predicate on `canonical_fips_xwalk` | **94 ms** |
|
||||
| no cohort filter at all (the floor) | 88 ms |
|
||||
## `complete = TRUE`: absent cells, labelled with why they are absent
|
||||
|
||||
**4.8x, within 7% of the floor.** The rendered `IN` list was 301,591
|
||||
characters and was re-parsed in 5-8 separate statements per call, so the cost
|
||||
was paid repeatedly; the predicate's size is constant in the cohort.
|
||||
* `cog_spending()` and `cog_revenue()` gain `complete`, defaulting to `FALSE`
|
||||
(today's behaviour). With `complete = TRUE` the requested grid is filled
|
||||
from the corpus's `code_set` table and every row carries a new
|
||||
`value_source` column:
|
||||
|
||||
`state` and `type` use the same vocabulary and the same internal coercion as
|
||||
`cog_gov_search()` -- `state` is a postal abbreviation (`"WI"`) even though the
|
||||
crosswalk column holds a FIPS code (`"55"`).
|
||||
| `value_source` | meaning | `amt_nominal` |
|
||||
|---|---|---|
|
||||
| `reported` | the corpus carries this cell | as published |
|
||||
| `census_zero` | dense-source year (≤ FY2011), cell absent — Census published `$0` | `0` |
|
||||
| `not_reported` | sparse-source year (≥ FY2012), cell absent — unknown | `NA` |
|
||||
|
||||
Supplying `govid` **and** `state`/`type` intersects them: the governments in
|
||||
`govid` that also match the predicate. Naming no cohort at all now aborts with
|
||||
class `uscogdata_no_cohort` rather than R's "argument is missing" error.
|
||||
The `NA` is deliberate and is the whole point: filling a modern absence
|
||||
with `0` would invent data, which is precisely the error the corpus's
|
||||
representation contract exists to prevent.
|
||||
* This restores information the reader lost when the corpus was sparsified
|
||||
(`SB194`, cog_pipeline#64) — a wide-era query whose cells were all `$0`
|
||||
had begun returning nothing at all — and improves on what came before it,
|
||||
since the pre-sparsification corpus could not distinguish a published zero
|
||||
from an unreported cell either.
|
||||
* The grid is scoped to each government's **own type**, so a county is never
|
||||
filled with cells only a state can report.
|
||||
* Needs a corpus published from 2026-07-29 onward (when `representation` and
|
||||
`code_set` began shipping); aborts with class
|
||||
`uscogdata_representation_unavailable` otherwise. Gated on the manifest
|
||||
listing those tables rather than on `schema_version`, which was never
|
||||
bumped for the change. Not available with `recipe` or
|
||||
`expenditure_concept = "total"` — neither draws its cells from `code_set`.
|
||||
* `provenance$completion` reports `applied`, `rows_filled`, and the per-year
|
||||
`absence_means` rule; `cog_explain()` prints a "Completion" section.
|
||||
|
||||
When the cohort is named by predicate there is no id list to report, so
|
||||
`provenance$scope$govids_found`/`govids_missing` are empty and
|
||||
`provenance$scope$cohort` carries `state`, `type` and `n_governments` instead.
|
||||
A `govid`-named cohort's provenance is unchanged.
|
||||
## Corpus-wide series breaks now reach users (`corpus_break_refs`)
|
||||
|
||||
## `cog_gov_search()` and `cog_balances()` gain `limit`/`offset`
|
||||
* Four catalogued series breaks carry `fin_code = "ALL"` — caveats about the
|
||||
corpus as a whole rather than about one item code. `series_break_refs` is
|
||||
built by matching `fin_code` against the item codes in the result, and no
|
||||
row's `item_code` is ever the literal `"ALL"`, so **none of them could ever
|
||||
be surfaced**: `SB085` (dollar precision across the 1976/1977 boundary),
|
||||
`SB087` (imputation exclusion from FY2002), `SB194` (the dense → sparse
|
||||
representation change at FY2012) and `SB086` (the government id scheme
|
||||
change at FY2017).
|
||||
* Provenance gains `corpus_break_refs`, selected on the break-year window
|
||||
alone and disjoint from `series_break_refs` by construction, so a consumer
|
||||
can tell a whole-result caveat from a break in one series. `cog_explain()`
|
||||
prints them under their own "Corpus-wide caveats" heading. cog-api passes
|
||||
provenance through verbatim, so the field appears there without an API
|
||||
change.
|
||||
* `SB194` is the one that made this urgent: a query spanning FY2011 → FY2012
|
||||
crosses the boundary where an absent cell stops meaning "Census published
|
||||
`$0`" and starts meaning "not reported", and until now nothing said so.
|
||||
|
||||
Pagination arrived on `cog_spending()`/`cog_revenue()` in 0.3.0; the other two
|
||||
verbs were left materializing everything and slicing in R. Both now take
|
||||
`limit`/`offset` with the same semantics: `NULL` default, the page applied in
|
||||
SQL behind a deterministic `ORDER BY`, and the unpaginated count returned as a
|
||||
`total_rows` attribute computed by `COUNT(*) OVER()` in the same scan rather
|
||||
than a second query.
|
||||
## Bundled fixture regenerated against the sparsified corpus
|
||||
|
||||
`cog_gov_search()` had no `LIMIT` at all, which made it the one verb that
|
||||
returns the entire 40,336-row crosswalk when called with no filter.
|
||||
* `inst/extdata/fixture_corpus/` now tracks the corpus published on
|
||||
2026-07-29 (`pipeline_commit 83f9715`, schema v6). The wide era no longer
|
||||
stores explicit zeros: FY2011 fell from 2,864,212 rows to 496,004, of
|
||||
which none are `$0`. **Absence now means two different things** — in a
|
||||
`dense_source` year (≤ FY2011) an absent cell means Census published `$0`;
|
||||
in a `sparse_source` year (≥ FY2012) it means not reported. The corpus
|
||||
carries that rule in two new tables the fixture now ships,
|
||||
`representation.parquet` and `code_set.parquet`, alongside
|
||||
`census_collection_coverage.parquet` and `lineage_events.parquet`
|
||||
(all ten publish-tree metadata tables, up from six). Catalogued upstream
|
||||
as series break `SB194`.
|
||||
* `cog_categories()` gains an `assistance` spending subtype: the J-prefix
|
||||
aid/benefit codes (`J19`, `J67`, `J68`, `J85`) are categorised now that
|
||||
the upstream crosswalk covers every flow code carrying dollars.
|
||||
* Two consequences worth knowing about, both visible in provenance rather
|
||||
than in returned dollars. The harmonization block's `na_rows_excluded`
|
||||
counts only rows that exist, so wide-era codes that were zero-padded no
|
||||
longer appear there. Coverage-gap `suggestions` are presence-based for the
|
||||
same reason, so a recipe whose component codes were all `$0` for a given
|
||||
government-year is no longer suggested for it.
|
||||
* `tests/testthat/test-fixture-vintage.R` pins these structural facts, so a
|
||||
fixture left behind by a future publish fails loudly instead of letting the
|
||||
suite pass against a corpus that no longer exists.
|
||||
|
||||
Two refusals rather than silent surprises:
|
||||
## Breaking: corpus schema_version 4 (Phase P canonical ids)
|
||||
|
||||
* `cog_balances(recipe = , limit = )` aborts with class
|
||||
`uscogdata_recipe_pagination_conflict` -- a recipe's result comes from a
|
||||
separate query that pagination is not wired into.
|
||||
* `cog_gov_search()` in basket mode (`length(name) > 1`) aborts with class
|
||||
`uscogdata_basket_pagination_conflict`. Basket mode returns one resolved row
|
||||
per requested name with a sidecar covering all of them; a page of that is not
|
||||
a page of anything the caller asked for.
|
||||
* The package now requires corpus `schema_version = 4` (`MinCorpusSchema` /
|
||||
`MaxCorpusSchema` in `DESCRIPTION` are both `4`); older corpora built
|
||||
against schema 3 are rejected by `cog_open()` with a clear version-mismatch
|
||||
error. `canonical_govid` is now uniformly 12 characters across every
|
||||
vintage the corpus covers (previously a mix of 9-char legacy ids and
|
||||
12-char FIPS ids depending on source year) — **every hardcoded
|
||||
`canonical_govid` literal from a pre-Phase-P corpus is now invalid** and
|
||||
must be re-resolved via `cog_gov_search()` or the new `canonical_alias`
|
||||
lookup table. `canonical_fips_xwalk` gains four columns
|
||||
(`legacy_govs_id`, `census_geoid`, `id_source`; `confidence` is renamed to
|
||||
`pop_confidence`) and a companion `canonical_alias` table ships in the
|
||||
corpus for mapping legacy/alternate ids onto the current canonical
|
||||
namespace. The bundled fixture corpus (`inst/extdata/fixture_corpus/`) has
|
||||
been regenerated against the Phase P publish tree, now ships the full
|
||||
`canonical_fips_xwalk` and `canonical_alias` master tables alongside the
|
||||
2019-2020 long partitions, and is reproducible via
|
||||
`data-raw/regenerate_fixture_corpus.R`.
|
||||
|
||||
## DuckDB's resource budget is configurable
|
||||
## Clearer errors when `USCOGDATA_URL` is unconfigured or returns non-JSON
|
||||
|
||||
`USCOGDATA_DUCKDB_THREADS` and `USCOGDATA_DUCKDB_MEMORY_LIMIT` (with matching
|
||||
`options(uscogdata.duckdb_threads = )` / `options(uscogdata.duckdb_memory_limit = )`
|
||||
spellings) cap the DuckDB connection the package opens. Both follow the same
|
||||
env-var > option > default precedence as `USCOGDATA_URL`.
|
||||
* `cog_open()` now aborts with the `uscogdata_url_not_configured` error
|
||||
class when the resolved corpus URL still contains the placeholder
|
||||
`REPLACE_WITH_SHARE_TOKEN` sentinel (or is empty). The message lists both
|
||||
remediation paths (`Sys.setenv(USCOGDATA_URL = ...)` and
|
||||
`options(uscogdata.url = ...)`) and points at the bundled fixture for
|
||||
offline testing. Previously the package proceeded to fetch the placeholder
|
||||
URL, cached the resulting HTML welcome page, and failed downstream with a
|
||||
cryptic `jsonlite` lexical-error.
|
||||
* `.fetch_or_cache_manifest()` now parses the HTTP response body before
|
||||
persisting it. Non-JSON responses (login pages, 404 HTML) raise
|
||||
`uscogdata_invalid_manifest` with the URL, Content-Type, and underlying
|
||||
parse error — and never write to the on-disk cache.
|
||||
* Manifest cache writes are now atomic (write to `manifest.json.tmp.<pid>`
|
||||
in `cache_dir`, then `file.rename` over the target), so an interrupted
|
||||
fetch cannot replace a previously-good cache.
|
||||
* Existing caches with non-JSON content (poisoned by the prior code path)
|
||||
are silently refetched instead of returning a parse error to the caller.
|
||||
* Local `USCOGDATA_URL` paths whose `manifest.json` is not valid JSON now
|
||||
surface the same `uscogdata_invalid_manifest` class with file context.
|
||||
|
||||
Unset, **no pragma is issued at all** and DuckDB's own defaults apply exactly as
|
||||
before -- every visible core. That is right for one interactive session on a
|
||||
dedicated machine and wrong for a server: where several readers share a host, each
|
||||
otherwise claims the whole machine and they contend. Capping measured ~5% on a
|
||||
single-government all-years query (502 ms at 2 threads vs 475 ms uncapped on 16
|
||||
cores), which is cheap enough that a server should always cap.
|
||||
## Per-capita denominators now use per-year Census F-33 population
|
||||
|
||||
This replaces a workaround in which a consumer reached into the package namespace
|
||||
at boot -- `getFromNamespace(".ensure_session", "uscogdata")()` followed by a manual
|
||||
`SET threads` -- depending both on a private name and on the session already being
|
||||
open.
|
||||
* `cog_spending()` and `cog_revenue()` previously divided all years' amounts
|
||||
by a single ACS 2018-2022 estimate (`canonical_fips_xwalk.population_acs`),
|
||||
producing biased per-capita values for time-series analysis. They now
|
||||
divide by the F-33 `population` recorded on each gov-year via the new
|
||||
`gov_population_yearly` view. Result tibbles gain a `pop_source` column
|
||||
with values `"census_f33"` or `"unavailable"`. `notes` is updated to
|
||||
concatenate multiple notes with `"; "`.
|
||||
|
||||
## Documentation: the corpus-access table is re-measured and honest
|
||||
## Peer cohorts can be set to a chosen year
|
||||
|
||||
The README's "two ways to read the corpus" table carried figures taken before
|
||||
the corpus was re-chunked into row groups (cog_pipeline#93, published
|
||||
2026-08-09) and reported the mirrored column as "local speed" with no number at
|
||||
all. Re-measured 2026-08-10 against the published corpus (`pipeline_commit
|
||||
3d28ddd`), fresh R session per arm:
|
||||
* `cog_find_peers()` adds a `year` argument (default: most recent year for
|
||||
which the target has an observed population in `gov_population_yearly`).
|
||||
The returned column previously named `population_acs` is now `population`
|
||||
and reflects the cohort year's vintage. The cohort year is attached to the
|
||||
returned tibble as `attr(x, "cohort_year")`.
|
||||
* `cog_peer_compare()` now stamps a `cohort_year` column on its result (read
|
||||
from the peers tibble's attribute) and records `cohort_year` plus
|
||||
`cohort_govids` in provenance. When the caller supplies a bare character
|
||||
vector instead of a `cog_find_peers()` result, `cohort_year` is `NA`.
|
||||
|
||||
* **A local mirror is roughly 60-80x faster.** A one-off question costs ~12 s
|
||||
end to end remotely against ~0.15 s mirrored. That is the largest single
|
||||
difference available to a user and it is now stated outright rather than left
|
||||
as "local speed".
|
||||
* **Opening the session is the largest remote cost** (~7.5 s -- manifest fetch
|
||||
plus 23 view registrations over HTTPS), larger than any individual query, and
|
||||
it lands on the first query rather than on `library(uscogdata)`. The old table
|
||||
did not account for it anywhere.
|
||||
* **The remote cost is round-trips, not scanning.** A repeat query over
|
||||
already-touched partitions is ~1.5 s against ~4 s cold, and a full-history
|
||||
query costs ~7 s whether it runs first or last.
|
||||
* The corpus size is **~201 MB**, not 190.6 MB -- row-group chunking added ~3.4%
|
||||
and the old figure was ambiguous between MB and MiB besides.
|
||||
* Documented that a burst of remote queries can be rate-limited by the host
|
||||
(`HTTP 429`), which is another reason to mirror for real work.
|
||||
## Rollups exclude govs missing population
|
||||
|
||||
## Fixes
|
||||
* `cog_geographic_rollup(per_capita = TRUE)` drops rows whose government has
|
||||
`pop_source == "unavailable"` and records the dropped govids in
|
||||
`provenance$rollup$excluded_govids`. This excludes special districts
|
||||
(type 4) and school districts (type 5) from per-capita rollups by design.
|
||||
|
||||
* `cog_gov_search()` now orders by `population_acs DESC NULLS LAST,
|
||||
canonical_govid`. **`population_acs` alone is not a total order** -- ties, and
|
||||
the entire `NULLS LAST` block, came back in whatever order the scan produced.
|
||||
That was invisible while every call returned the full result set, but it makes
|
||||
a paged sweep unsound: two requests can order tied rows differently, so a row
|
||||
is duplicated on one page and missing from the next. Unpaginated results are
|
||||
unchanged except for the relative order of rows that were already tied.
|
||||
## New: vignette and provenance metadata
|
||||
|
||||
* An unknown `state` abbreviation now aborts with "Unknown state abbreviation"
|
||||
(class `uscogdata_unknown_state`) instead of base R's "subscript out of
|
||||
bounds". `.state_abbrev_to_fips` is a named character vector, so `[[` on an
|
||||
absent name threw before the curated message could be reached -- making that
|
||||
message unreachable dead code in every verb that takes a `state`.
|
||||
|
||||
# uscogdata 0.3.0
|
||||
|
||||
First public release.
|
||||
|
||||
`uscogdata` provides curated R verbs over the Civilytics US Census of
|
||||
Governments finance corpus: unit-level financial profiles, geographic rollups
|
||||
and peer comparisons, with auditable provenance on every result.
|
||||
|
||||
## What it covers
|
||||
|
||||
Government types 0-3 (state, county, municipality, township), FY1967-FY2024 --
|
||||
56 fiscal years, 46,148,034 rows, 190.6 MB. There is no source data for FY1968
|
||||
or FY1969. Special districts (type 4) and school districts (type 5) are out of
|
||||
scope pending validation.
|
||||
|
||||
## The verbs
|
||||
|
||||
`cog_spending()`, `cog_revenue()` and `cog_balances()` for flows and holdings;
|
||||
`cog_gov_search()` to resolve place names (including basket mode for many at
|
||||
once); `cog_find_peers()` and `cog_peer_compare()` for cohorts;
|
||||
`cog_geographic_rollup()` for aggregates; `cog_categories()`, `cog_recipes()`,
|
||||
`cog_manifest()` and `cog_explain()` for metadata and provenance; and
|
||||
`cog_mirror()` for a local copy of the corpus.
|
||||
|
||||
## Reading the corpus now works out of the box
|
||||
|
||||
* The package reads the published corpus over HTTPS **with no configuration**.
|
||||
Previously the default was a placeholder sentinel and no document in the
|
||||
package supplied a working URL, so a new user had no path to a session.
|
||||
* Remote reads work at all. The partitioned view used a glob, and DuckDB
|
||||
cannot expand a glob over generic HTTP -- there is no directory listing to
|
||||
expand against. Partition paths are now enumerated from the corpus manifest,
|
||||
which is host-agnostic: an HTTPS mirror, a Nextcloud share and a local
|
||||
`cog_mirror()` copy all take the same path.
|
||||
* Nothing is written to disk in remote mode; DuckDB fetches only the row
|
||||
groups a query needs.
|
||||
|
||||
## Four things to know before your first query
|
||||
|
||||
* **Amounts are in full US dollars.** The raw Census files report thousands;
|
||||
the verbs multiply by 1000 on the way out. Do not multiply again.
|
||||
* **Multi-government aggregates disclose their coverage.** The Census is a
|
||||
complete enumeration only in years ending in 2 and 7; every other year is a
|
||||
sample. Every such result carries `provenance$coverage` with per-year
|
||||
`n_units_reporting`.
|
||||
* **Absence means two different things.** Before FY2012 an absent cell means
|
||||
Census published $0; from FY2012 it means not reported. `complete = TRUE`
|
||||
labels which.
|
||||
* **Series breaks reach you unasked.** Catalogued breaks intersecting your
|
||||
query appear in provenance and in `cog_explain()`.
|
||||
|
||||
## Known limits
|
||||
|
||||
* Special districts (type 4) and school districts (type 5) are out of scope.
|
||||
* Per-capita rollups exclude governments with no F-33 population, which is by
|
||||
design but does silently narrow a rollup.
|
||||
* `n_units_reporting` is category-conditional and is not a response rate.
|
||||
* Employee-retirement (`X`) codes stop at FY2016, when those systems moved to
|
||||
the Annual Survey of Public Pensions.
|
||||
|
||||
# uscogdata 0.2.0
|
||||
* New vignette `population-denominators` covers the four population sources,
|
||||
the type-4/5 coverage gap, the popyear quirk, and how to build moving-window
|
||||
peer cohorts manually.
|
||||
* Provenance gains `transformations$per_capita$popyear_range` and
|
||||
`pop_source_counts`. `cog_explain()` renders both.
|
||||
|
||||
## New features
|
||||
|
||||
* `cog_spending()` and `cog_revenue()` accept the reserved category
|
||||
`"All Categories"`, returning one summed row per
|
||||
`(year, canonical_govid, subtype)` across every category inside the
|
||||
requested concept's subtype scope. Filtering the result to
|
||||
`spend_subtype == "operations"` gives an operating-expenditure total.
|
||||
`cog_geographic_rollup()` inherits it,
|
||||
which is the efficient way to build a geographic total — previously a
|
||||
caller had to issue one rollup per category and sum the results
|
||||
(cog-api#37).
|
||||
* `cog_gov_search()` gains a **basket mode**: passing vector `name`
|
||||
/ `state` / `type` arguments resolves multiple place names in one
|
||||
call and returns a tibble of canonical rows in input order, ready
|
||||
to pipe into `cog_spending()` / `cog_revenue()`. Per-row resolution
|
||||
follows an exact-then-substring matching algorithm with deterministic
|
||||
disambiguation; ambiguous and missing entries are surfaced via a
|
||||
sidecar audit tibble plus a single console summary message.
|
||||
* New exports `cog_basket_resolution()` and `cog_basket_unresolved()`
|
||||
expose the basket sidecar for iterative query refinement.
|
||||
|
||||
`"All Categories"` is not the same thing as `expenditure_concept = "total"`.
|
||||
The concept chooses which subtypes are in scope; `"All Categories"` chooses
|
||||
whether the rows inside that scope are broken out or summed.
|
||||
## Breaking changes
|
||||
|
||||
* `cog_categories()` advertises `"All Categories"` for the expenditure and
|
||||
revenue vocabularies, so the reserved value is discoverable.
|
||||
|
||||
* Coverage signposting (see "Signposting now catches partially-suppressed
|
||||
categories" below) now also works in `category = "All Categories"` mode.
|
||||
The recipe-suggestion candidate query used to be scoped by `category`,
|
||||
which is never a match for the reserved `"All Categories"` value, so
|
||||
`provenance$suggestions` always came back empty there — the one mode whose
|
||||
whole point is "you cannot sum the wrong scope" was silently unable to
|
||||
signal a wrong scope. The candidate query is now scoped by the concept's
|
||||
subtype allowlist instead, symmetric with how `.build_verb_sql()` itself
|
||||
scopes the summed total: Los Angeles County FY2011, `category = "All
|
||||
Categories"` still excludes $271,589,000 of aggregate-published Public
|
||||
Welfare (`E68`), but now names `recipe = "welfare_cash_e68_wide"` to
|
||||
recover it instead of reporting zero suggestions.
|
||||
|
||||
## Documentation
|
||||
|
||||
* `cog_geographic_rollup()` and `cog_peer_compare()` now document that
|
||||
`provenance$coverage`'s `n_units_reporting` is **category-conditional** and
|
||||
is not a response rate: a government that was surveyed and genuinely spends
|
||||
nothing in the requested category is indistinguishable from one never
|
||||
surveyed (uscogdata#36).
|
||||
* The first formal of `cog_gov_search()` was renamed from `pattern`
|
||||
to `name`. All existing call sites in `cog_explorer/` and the
|
||||
package itself use positional first-arg, so this rename is
|
||||
non-breaking in practice. Callers that pass `pattern = ...` by name
|
||||
must update to `name = ...`.
|
||||
|
||||
+9
-66
@@ -18,9 +18,7 @@
|
||||
#' comparable to a GAAP fund balance from an ACFR.
|
||||
#'
|
||||
#' @param govid Canonical govid(s): a character vector, or a data frame with a
|
||||
#' `canonical_govid` column (e.g. from [cog_gov_search()]). `NULL` to name
|
||||
#' the cohort by `state`/`type` instead.
|
||||
#' @inheritParams cog_spending
|
||||
#' `canonical_govid` column (e.g. from [cog_gov_search()]).
|
||||
#' @param years Integer vector of fiscal years.
|
||||
#' @param category Optional character vector of categories to keep. One of
|
||||
#' `"Fund Balances"`, `"Insurance Trust Balances"`,
|
||||
@@ -30,12 +28,7 @@
|
||||
#' every combination would be either redundant or empty.
|
||||
#' `category = "Fund Balances"` is exactly the `general` family
|
||||
#' (`W01`/`W31`/`W61`). `balance_subtype` is returned, so a finer split is
|
||||
#' one `dplyr::filter()` away. The reserved pseudo-category
|
||||
#' `"All Categories"` (see [cog_spending()]) is **not** supported here and
|
||||
#' errors with class `uscogdata_all_categories_unsupported`: it sums a
|
||||
#' concept's subtype scope, and holdings are a stock with no concept
|
||||
#' vocabulary to sum across. Omit `category` to get every category broken
|
||||
#' out instead.
|
||||
#' one `dplyr::filter()` away.
|
||||
#' @param per_capita Divide holdings by population. Note this is a **stock per
|
||||
#' resident** (reserves per person), which is *not* comparable to
|
||||
#' [cog_spending()]'s per-capita figures -- those are a flow per person.
|
||||
@@ -47,12 +40,6 @@
|
||||
#' @param recipe Optional harmonization recipe id (see [cog_recipes()]).
|
||||
#' `"cash_securities_z77_wide"` and `"cash_securities_z78_wide"` bridge the
|
||||
#' wide era to the modern one.
|
||||
#' @param limit Maximum number of result rows to return, pushed into the SQL
|
||||
#' rather than applied after materializing every row. `NULL` (default)
|
||||
#' returns everything. Cannot be combined with `recipe` -- see `offset` and
|
||||
#' `total_rows`.
|
||||
#' @param offset Rows to skip before `limit` starts counting (0-based).
|
||||
#' Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set.
|
||||
#'
|
||||
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||
#' `balance_subtype`, `category`, `amt_nominal`, `codes_included`,
|
||||
@@ -70,22 +57,16 @@
|
||||
#' and `truncated` (the observed subtypes whose coverage falls short of the
|
||||
#' requested years). `expenditure_concept`/`revenue_concept` are `NA` --
|
||||
#' holdings are a stock, not a flow, so neither concept vocabulary applies.
|
||||
#'
|
||||
#' When `limit` is set, also carries a `total_rows` attribute: the full
|
||||
#' unpaginated row count, computed by the same query (`COUNT(*) OVER()`)
|
||||
#' rather than a second scan.
|
||||
#' @export
|
||||
cog_balances <- function(govid = NULL, years, category = NULL,
|
||||
cog_balances <- function(govid, years, category = NULL,
|
||||
per_capita = FALSE, adjust_to_year = NULL,
|
||||
basis = c("harmonized", "raw"), recipe = NULL,
|
||||
state = NULL, type = NULL,
|
||||
limit = NULL, offset = NULL) {
|
||||
basis = c("harmonized", "raw"), recipe = NULL) {
|
||||
call <- match.call()
|
||||
basis <- match.arg(basis, c("harmonized", "raw"))
|
||||
# Coerce FIRST, validate second: .validate_verb_inputs() asserts
|
||||
# is.character(govid), and a data-frame govid (cog_gov_search() output) has
|
||||
# not been unwrapped yet at this point.
|
||||
govid <- if (is.null(govid)) NULL else .coerce_govid_input(govid)
|
||||
govid <- .coerce_govid_input(govid)
|
||||
# The money verbs' validator, reused rather than re-implemented (R/spending.R).
|
||||
# It covers the exact superset cog_balances() needs -- including the
|
||||
# recipe/category mutual-exclusivity guard -- so a second local copy would
|
||||
@@ -93,36 +74,11 @@ cog_balances <- function(govid = NULL, years, category = NULL,
|
||||
# helper reuse as .build_verb_sql()/.attach_per_capita() below; it does NOT
|
||||
# route the verb through .verb_spendrev(), which stays deliberately unused
|
||||
# here because its flow vocabulary is meaningless for a stock.
|
||||
#
|
||||
# allow_all_categories is left at its FALSE default (contrast
|
||||
# .verb_spendrev(), which passes TRUE): the all-categories mode's "sum"
|
||||
# only means something in terms of a concept's subtype scope, and holdings
|
||||
# have no concept vocabulary. The reuse above is exactly why this can be a
|
||||
# one-line default rather than a second bespoke check -- see the
|
||||
# validator's own doc comment for the incident that made that matter.
|
||||
.validate_verb_inputs(govid, years, category, per_capita, adjust_to_year,
|
||||
recipe)
|
||||
|
||||
# Same semantics as the money verbs (R/pagination.R). Only the `recipe`
|
||||
# conflict applies here: cog_balances() has no `complete` argument, and a
|
||||
# recipe's result comes from .run_recipe()'s own query, which pagination is
|
||||
# not wired into.
|
||||
paging <- .validate_pagination(limit, offset)
|
||||
limit <- paging$limit
|
||||
offset <- paging$offset
|
||||
if (!is.null(limit) && !is.null(recipe)) {
|
||||
cli::cli_abort(c(
|
||||
"`limit`/`offset` cannot be combined with `recipe`.",
|
||||
"i" = "A recipe's result comes from a separate query (`.run_recipe()`) that pagination is not wired into yet.",
|
||||
"*" = "Drop `limit`/`offset`, or drop `recipe`."
|
||||
), class = "uscogdata_recipe_pagination_conflict")
|
||||
}
|
||||
|
||||
years <- as.integer(years)
|
||||
if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year)
|
||||
|
||||
cohort <- .make_cohort(govid, state, type)
|
||||
|
||||
con <- .ensure_session()
|
||||
.require_balance_support(con)
|
||||
scope <- .check_govids_in_scope(govid)
|
||||
@@ -135,14 +91,13 @@ cog_balances <- function(govid = NULL, years, category = NULL,
|
||||
manifest <- .uscogdata_env$manifest
|
||||
recipe_block <- NULL
|
||||
category_for_prov <- category
|
||||
total_rows <- NULL # set below only when limit is non-NULL (non-recipe path)
|
||||
|
||||
if (!is.null(recipe)) {
|
||||
.require_schema_v5(con, manifest, "recipe =")
|
||||
.validate_recipe_id(con, recipe)
|
||||
comps <- .recipe_components(con, recipe)
|
||||
recipe_label <- comps$label[[1]]
|
||||
result <- .run_recipe(con, recipe, cohort, years)
|
||||
result <- .run_recipe(con, recipe, govid, years)
|
||||
sql <- attr(result, "sql_query")
|
||||
result <- .shape_recipe_result(result, "balance_subtype", recipe_label)
|
||||
recipe_block <- list(
|
||||
@@ -152,25 +107,15 @@ cog_balances <- function(govid = NULL, years, category = NULL,
|
||||
category_for_prov <- recipe_label
|
||||
} else {
|
||||
sql <- .build_verb_sql("balance_annotated", "balance_subtype",
|
||||
cohort, years, category,
|
||||
ig_view = NULL, subtype_scope = NULL,
|
||||
limit = limit, offset = offset)
|
||||
govid, years, category,
|
||||
ig_view = NULL, subtype_scope = NULL)
|
||||
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
if (!is.null(limit)) {
|
||||
paged <- .take_pagination_total(result, con, function() {
|
||||
.build_verb_sql("balance_annotated", "balance_subtype",
|
||||
cohort, years, category,
|
||||
ig_view = NULL, subtype_scope = NULL)
|
||||
})
|
||||
result <- paged$result
|
||||
total_rows <- paged$total_rows
|
||||
}
|
||||
}
|
||||
|
||||
# Order matters (matches .verb_spendrev()): per-capita first, so
|
||||
# .attach_real_dollars() deflates the nominal per-capita column into
|
||||
# amt_per_capita_real rather than needing amt_per_capita_nominal recomputed.
|
||||
if (isTRUE(per_capita)) result <- .attach_per_capita(result, con)
|
||||
if (isTRUE(per_capita)) result <- .attach_per_capita(result, con, govid)
|
||||
if (!is.null(adjust_to_year)) {
|
||||
result <- .attach_real_dollars(result, adjust_to_year, per_capita)
|
||||
}
|
||||
@@ -188,7 +133,6 @@ cog_balances <- function(govid = NULL, years, category = NULL,
|
||||
)
|
||||
prov$scope$govids_found <- scope$found
|
||||
prov$scope$govids_missing <- scope$missing
|
||||
prov$scope$cohort <- .cohort_provenance(con, cohort)
|
||||
|
||||
prov$balance_caveats <- .balance_caveats(
|
||||
con, prov$codes_summed$observed, years
|
||||
@@ -196,7 +140,6 @@ cog_balances <- function(govid = NULL, years, category = NULL,
|
||||
.emit_balance_caveats(prov$balance_caveats)
|
||||
|
||||
attr(result, "provenance") <- prov
|
||||
if (!is.null(limit)) attr(result, "total_rows") <- total_rows
|
||||
result
|
||||
}
|
||||
|
||||
|
||||
@@ -54,7 +54,7 @@
|
||||
#' expenditure_concept = "total": ig_long_harmonized COALESCEs rather than
|
||||
#' drops NULL-harmonized rows, so harmonization never excludes an IG row.
|
||||
#' @noRd
|
||||
.build_harmonization_block <- function(con, cohort, years, resolved,
|
||||
.build_harmonization_block <- function(con, govid, years, resolved,
|
||||
subtype_col, subtype_scope) {
|
||||
if (!identical(resolved$basis, "harmonized")) {
|
||||
return(list(
|
||||
@@ -68,12 +68,12 @@
|
||||
sql <- sprintf(
|
||||
"SELECT COUNT(*) AS n, COALESCE(SUM(amt), 0) * 1000.0 AS amt
|
||||
FROM long
|
||||
WHERE %s AND year IN (%s)
|
||||
WHERE canonical_govid IN (%s) AND year IN (%s)
|
||||
AND NOT is_aggregate AND harmonized_code IS NULL
|
||||
AND item_code IN (
|
||||
SELECT item_code FROM summary_categories WHERE %s IN (%s)
|
||||
)",
|
||||
.cohort_sql(cohort), paste(as.integer(years), collapse = ","),
|
||||
.sql_lit_chr(govid), paste(as.integer(years), collapse = ","),
|
||||
subtype_col, .sql_lit_chr(subtype_scope)
|
||||
)
|
||||
na <- DBI::dbGetQuery(con, sql)
|
||||
|
||||
+2
-28
@@ -22,10 +22,7 @@
|
||||
#' `category` column (e.g. `"Police"` or `"Tax"`).
|
||||
#' @return Tibble with columns `category`, `category_type`, `subtype`,
|
||||
#' `n_codes`, `item_codes` (comma-separated, alphabetical). Sorted by
|
||||
#' `category_type`, `category`, `subtype`. Includes one row per flow for the
|
||||
#' reserved pseudo-category `"All Categories"`, which carries `NA` for
|
||||
#' `subtype`, `n_codes` and `item_codes` because it is a query mode rather
|
||||
#' than a crosswalk entry — see [cog_spending()]'s `category` argument.
|
||||
#' `category_type`, `category`, `subtype`.
|
||||
#' @export
|
||||
cog_categories <- function(type = NULL, pattern = NULL) {
|
||||
if (!is.null(type)) {
|
||||
@@ -66,28 +63,5 @@ cog_categories <- function(type = NULL, pattern = NULL) {
|
||||
"GROUP BY category, category_type, subtype
|
||||
ORDER BY category_type, category, subtype"
|
||||
)
|
||||
out <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
|
||||
# The reserved pseudo-category is a query mode, not a crosswalk row, so it
|
||||
# has no item codes to report -- hence NA rather than 0 for n_codes. It is
|
||||
# emitted for the two FLOW vocabularies only: cog_balances() returns a stock
|
||||
# and has no concept argument to sum within.
|
||||
pseudo <- tibble::tibble(
|
||||
category = .ALL_CATEGORIES,
|
||||
category_type = c("expenditure", "revenue"),
|
||||
subtype = NA_character_,
|
||||
n_codes = NA_integer_,
|
||||
item_codes = NA_character_
|
||||
)
|
||||
if (!is.null(type)) {
|
||||
db_type <- if (type == "spending") "expenditure" else type
|
||||
pseudo <- pseudo[pseudo$category_type == db_type, , drop = FALSE]
|
||||
}
|
||||
if (!is.null(pattern) && nrow(pseudo) > 0L) {
|
||||
keep <- grepl(pattern, pseudo$category, ignore.case = TRUE)
|
||||
pseudo <- pseudo[keep, , drop = FALSE]
|
||||
}
|
||||
if (nrow(pseudo) == 0L) return(out)
|
||||
out <- rbind(out, pseudo)
|
||||
out[order(out$category_type, out$category, out$subtype), , drop = FALSE]
|
||||
tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
}
|
||||
|
||||
-127
@@ -1,127 +0,0 @@
|
||||
# How a verb names the set of governments it queries.
|
||||
#
|
||||
# Historically there was one way: a `govid` character vector, rendered by
|
||||
# .sql_lit_chr() into a quoted IN list. That is fine for a handful of
|
||||
# governments and pathological for a fleet. Measured against the production
|
||||
# corpus, the same FY2022 aggregate over the 20,106-government `type = "city"`
|
||||
# cohort:
|
||||
#
|
||||
# cohort expressed as time
|
||||
# IN (20,106 literals) 449 ms
|
||||
# join against a temp cohort table 99 ms
|
||||
# predicate on canonical_fips_xwalk 94 ms
|
||||
# no cohort filter at all (the floor) 88 ms
|
||||
#
|
||||
# 4.8x, and within 7% of the no-filter floor. The rendered IN list is 301,591
|
||||
# characters and .verb_spendrev() embeds it in 5-8 separate statements per
|
||||
# call, so the parse-and-plan cost is paid over and over (uscogdata#58).
|
||||
#
|
||||
# A cohort therefore has two independent halves, and a query can carry either
|
||||
# or both:
|
||||
#
|
||||
# ids an explicit canonical_govid vector -> literal IN list
|
||||
# predicate state/type over canonical_fips_xwalk -> IN (SELECT ...)
|
||||
#
|
||||
# Both together is an INTERSECTION -- "these ids, narrowed to that state/type"
|
||||
# -- never a precedence rule where one silently wins.
|
||||
|
||||
#' Build the internal cohort object shared by every query verb.
|
||||
#'
|
||||
#' `state` and `type` are coerced with the SAME helpers `cog_gov_search()`
|
||||
#' uses. That is load-bearing, not tidiness: the public argument is a postal
|
||||
#' abbreviation (`"WI"`) while `canonical_fips_xwalk.fips_state` holds a FIPS
|
||||
#' code (`"55"`), and `type` is a label (`"city"`) against an integer
|
||||
#' `govs_type`. A predicate written against the raw parameter matches nothing
|
||||
#' and returns an empty result indistinguishable from "this government
|
||||
#' reported nothing" -- cog-api hit exactly that trap optimizing this path.
|
||||
#' One definition of the translation, not two.
|
||||
#'
|
||||
#' @param govid Already-coerced character vector of canonical_govids, or NULL.
|
||||
#' @param state Postal abbreviation or FIPS code, or NULL.
|
||||
#' @param type Type label or integer code, or NULL.
|
||||
#' @noRd
|
||||
.make_cohort <- function(govid = NULL, state = NULL, type = NULL) {
|
||||
if (is.null(govid) && is.null(state) && is.null(type)) {
|
||||
cli::cli_abort(c(
|
||||
"A cohort must be named.",
|
||||
"*" = "Pass {.arg govid} for specific governments, or {.arg state}/{.arg type} for every government matching a predicate.",
|
||||
"i" = "Passing both intersects them: the governments in {.arg govid} that also match {.arg state}/{.arg type}."
|
||||
), class = "uscogdata_no_cohort")
|
||||
}
|
||||
structure(
|
||||
list(
|
||||
ids = govid,
|
||||
state = state,
|
||||
type = type,
|
||||
state_fips = if (is.null(state)) NULL else .coerce_state_to_fips(state),
|
||||
type_int = if (is.null(type)) NULL else .coerce_type(type)
|
||||
),
|
||||
class = "uscogdata_cohort"
|
||||
)
|
||||
}
|
||||
|
||||
#' Is any part of this cohort expressed as an xwalk predicate?
|
||||
#' @noRd
|
||||
.cohort_by_predicate <- function(cohort) {
|
||||
!is.null(cohort$state_fips) || !is.null(cohort$type_int)
|
||||
}
|
||||
|
||||
#' Render the cohort as a SQL boolean expression over `col`.
|
||||
#'
|
||||
#' `col` may be qualified (`"l.canonical_govid"`, `"x.canonical_govid"`) --
|
||||
#' several call sites join the xwalk under an alias. The subquery's own
|
||||
#' projected column stays unqualified: it selects from canonical_fips_xwalk,
|
||||
#' not from the outer relation.
|
||||
#' @noRd
|
||||
.cohort_sql <- function(cohort, col = "canonical_govid") {
|
||||
preds <- character(0)
|
||||
|
||||
if (!is.null(cohort$ids)) {
|
||||
preds <- c(preds, sprintf("%s IN (%s)", col, .sql_lit_chr(cohort$ids)))
|
||||
}
|
||||
|
||||
if (.cohort_by_predicate(cohort)) {
|
||||
xwalk_preds <- character(0)
|
||||
if (!is.null(cohort$state_fips)) {
|
||||
xwalk_preds <- c(xwalk_preds,
|
||||
sprintf("fips_state = %s", .sql_lit_chr(cohort$state_fips)))
|
||||
}
|
||||
if (!is.null(cohort$type_int)) {
|
||||
xwalk_preds <- c(xwalk_preds, sprintf("govs_type = %d", cohort$type_int))
|
||||
}
|
||||
preds <- c(preds, sprintf(
|
||||
"%s IN (SELECT canonical_govid FROM canonical_fips_xwalk WHERE %s)",
|
||||
col, paste(xwalk_preds, collapse = " AND ")
|
||||
))
|
||||
}
|
||||
|
||||
paste(preds, collapse = " AND ")
|
||||
}
|
||||
|
||||
#' How many governments the cohort covers.
|
||||
#'
|
||||
#' One COUNT against the crosswalk, used only to populate the provenance
|
||||
#' `scope$cohort` block. Deliberately a count rather than the id list: a
|
||||
#' fleet-scale cohort would otherwise put 20,000 ids into every response body,
|
||||
#' which is the cost this issue exists to remove.
|
||||
#' @noRd
|
||||
.cohort_count <- function(con, cohort) {
|
||||
sql <- sprintf(
|
||||
"SELECT COUNT(*) AS n FROM canonical_fips_xwalk WHERE %s",
|
||||
.cohort_sql(cohort)
|
||||
)
|
||||
as.integer(DBI::dbGetQuery(con, sql)$n[[1]])
|
||||
}
|
||||
|
||||
#' The provenance `scope$cohort` block for a predicate cohort, or NULL when
|
||||
#' the cohort was named by id alone (in which case `govids_found`/
|
||||
#' `govids_missing` already describe it exactly).
|
||||
#' @noRd
|
||||
.cohort_provenance <- function(con, cohort) {
|
||||
if (!.cohort_by_predicate(cohort)) return(NULL)
|
||||
list(
|
||||
state = if (is.null(cohort$state)) NA_character_ else as.character(cohort$state),
|
||||
type = if (is.null(cohort$type)) NA_character_ else as.character(cohort$type),
|
||||
n_governments = .cohort_count(con, cohort)
|
||||
)
|
||||
}
|
||||
+5
-5
@@ -57,7 +57,7 @@
|
||||
#' aggregate rows. Without it the grid would offer cells the verb structurally
|
||||
#' never returns, so every one of them would fill as a phantom $0.
|
||||
#' @noRd
|
||||
.completion_grid_sql <- function(subtype_col, cohort, years, category,
|
||||
.completion_grid_sql <- function(subtype_col, govid, years, category,
|
||||
subtype_scope) {
|
||||
category_pred <- if (is.null(category)) {
|
||||
""
|
||||
@@ -76,13 +76,13 @@
|
||||
JOIN canonical_fips_xwalk x ON x.govs_type = cs.type
|
||||
JOIN summary_categories c ON c.item_code = cs.item_code
|
||||
JOIN representation r ON r.year = cs.year
|
||||
WHERE %2$s
|
||||
WHERE x.canonical_govid IN (%2$s)
|
||||
AND cs.year IN (%3$s)
|
||||
AND NOT cs.is_aggregate
|
||||
AND c.category IS NOT NULL
|
||||
AND c.%1$s IN (%4$s)
|
||||
%5$s",
|
||||
subtype_col, .cohort_sql(cohort, "x.canonical_govid"),
|
||||
subtype_col, .sql_lit_chr(govid),
|
||||
paste(as.integer(years), collapse = ","),
|
||||
.sql_lit_chr(subtype_scope), category_pred
|
||||
)
|
||||
@@ -94,10 +94,10 @@
|
||||
#' provenance block. Reported rows are passed through untouched -- filling
|
||||
#' must never alter or drop what the corpus actually published.
|
||||
#' @noRd
|
||||
.complete_result <- function(result, con, subtype_col, cohort, years, category,
|
||||
.complete_result <- function(result, con, subtype_col, govid, years, category,
|
||||
subtype_scope) {
|
||||
grid <- tibble::as_tibble(DBI::dbGetQuery(
|
||||
con, .completion_grid_sql(subtype_col, cohort, years, category, subtype_scope)
|
||||
con, .completion_grid_sql(subtype_col, govid, years, category, subtype_scope)
|
||||
))
|
||||
|
||||
result$value_source <- rep("reported", nrow(result))
|
||||
|
||||
+2
-72
@@ -5,28 +5,9 @@
|
||||
.uscogdata_env <- new.env(parent = emptyenv())
|
||||
|
||||
.uscogdata_defaults <- list(
|
||||
# Public HuggingFace mirror of the published corpus: CC-BY-4.0, no
|
||||
# credential, CDN-backed. This is the default so `library(uscogdata)`
|
||||
# followed by a verb works with zero configuration -- previously the
|
||||
# default was a REPLACE_WITH_SHARE_TOKEN sentinel and no document in the
|
||||
# package supplied a working URL, so a new user had no path to a session.
|
||||
#
|
||||
# The trailing slash is required: every consumer concatenates onto this
|
||||
# (see .resolve_url(), which enforces it anyway).
|
||||
#
|
||||
# Override with USCOGDATA_URL or options(uscogdata.url=) to read a
|
||||
# Nextcloud share or a local copy made by cog_mirror().
|
||||
url = "https://huggingface.co/datasets/civilytics/us-cog-finance/resolve/main/",
|
||||
url = "https://cloud.civilytics.org/s/REPLACE_WITH_SHARE_TOKEN/download/",
|
||||
cache_dir = NULL,
|
||||
manifest_ttl_secs = 3600L,
|
||||
# NULL means "emit no pragma", which leaves DuckDB's own defaults intact:
|
||||
# every visible core, and 80% of RAM. That is right for one interactive
|
||||
# session on a dedicated machine and wrong for a server, where several
|
||||
# readers share a box and each would otherwise claim all of it. See
|
||||
# .resolve_duckdb_threads() for why this is a supported option rather than
|
||||
# something a consumer reaches into the namespace to set.
|
||||
duckdb_threads = NULL,
|
||||
duckdb_memory_limit = NULL
|
||||
manifest_ttl_secs = 3600L
|
||||
)
|
||||
|
||||
#' Resolve a config value: env var > option > default
|
||||
@@ -69,54 +50,3 @@
|
||||
v <- .cfg("cache_dir")
|
||||
if (is.null(v)) tools::R_user_dir("uscogdata", "cache") else v
|
||||
}
|
||||
|
||||
#' Resolve the DuckDB thread cap, or NULL to leave DuckDB's default alone.
|
||||
#'
|
||||
#' `cog_open()` used to connect with a bare `dbConnect()` and set no `threads`
|
||||
#' pragma, so DuckDB claimed every core it could see. cog-api works around that
|
||||
#' by reaching into this namespace at boot --
|
||||
#' `getFromNamespace(".ensure_session", "uscogdata")()` followed by a manual
|
||||
#' `SET threads` -- which depends on a private name and on the session already
|
||||
#' being open. Making it a resolved option removes the reason to do that.
|
||||
#'
|
||||
#' `.cfg()` returns an environment variable as CHARACTER, so this coerces
|
||||
#' rather than trusting the type: `USCOGDATA_DUCKDB_THREADS=4` arrives as "4",
|
||||
#' and `sprintf("SET threads TO %d", "4")` would abort inside the connection
|
||||
#' path with an error about the pragma rather than about the setting.
|
||||
#' @noRd
|
||||
.resolve_duckdb_threads <- function() {
|
||||
v <- .cfg("duckdb_threads")
|
||||
if (is.null(v) || (is.character(v) && !nzchar(v))) return(NULL)
|
||||
n <- suppressWarnings(as.integer(v))
|
||||
if (length(n) != 1L || is.na(n) || n < 1L) {
|
||||
cli::cli_abort(c(
|
||||
"{.envvar USCOGDATA_DUCKDB_THREADS} must be a single positive integer.",
|
||||
x = "Got {.val {v}}.",
|
||||
i = "Unset it (or {.code options(uscogdata.duckdb_threads = NULL)}) to use DuckDB's default of every visible core."
|
||||
), class = "uscogdata_invalid_duckdb_threads")
|
||||
}
|
||||
n
|
||||
}
|
||||
|
||||
#' Resolve the DuckDB memory limit, or NULL to leave DuckDB's default alone.
|
||||
#'
|
||||
#' The value is a DuckDB size string (`"4GB"`, `"512MB"`). Only its SHAPE is
|
||||
#' checked here -- DuckDB owns the unit vocabulary, and re-implementing that
|
||||
#' parse would be a second definition free to drift from the engine's. An
|
||||
#' unrecognised unit therefore surfaces as DuckDB's own error at `SET` time,
|
||||
#' which names the setting correctly; the check here exists to reject the
|
||||
#' inputs that would otherwise reach the connection as a SQL fragment.
|
||||
#' @noRd
|
||||
.resolve_duckdb_memory_limit <- function() {
|
||||
v <- .cfg("duckdb_memory_limit")
|
||||
if (is.null(v) || (is.character(v) && !nzchar(v))) return(NULL)
|
||||
if (length(v) != 1L || !is.character(v) ||
|
||||
!grepl("^[0-9]+(\\.[0-9]+)?\\s*[A-Za-z]{0,3}$", v)) {
|
||||
cli::cli_abort(c(
|
||||
"{.envvar USCOGDATA_DUCKDB_MEMORY_LIMIT} must be a single DuckDB size string.",
|
||||
x = "Got {.val {v}}.",
|
||||
i = "Examples: {.val 4GB}, {.val 512MB}, {.val 1.5GB}."
|
||||
), class = "uscogdata_invalid_duckdb_memory_limit")
|
||||
}
|
||||
trimws(v)
|
||||
}
|
||||
|
||||
+1
-32
@@ -11,30 +11,6 @@
|
||||
#' returns `result` invisibly for chaining. `"list"` returns the raw
|
||||
#' provenance list (identical to `attr(result, "provenance")`).
|
||||
#' @return Either `result` (invisibly) or the provenance list.
|
||||
#' @section Two kinds of series break:
|
||||
#' Catalogued breaks reach you without being asked for, in two disjoint
|
||||
#' fields, because a caveat about one series and a caveat about the whole
|
||||
#' corpus are different claims:
|
||||
#'
|
||||
#' * **`series_break_refs`** — breaks matched against the item codes actually
|
||||
#' present in this result. A break in one code you queried.
|
||||
#' * **`corpus_break_refs`** — breaks catalogued with `fin_code = "ALL"`,
|
||||
#' which are statements about the corpus rather than about any one code:
|
||||
#' dollar precision across the 1976/1977 boundary (`SB085`), imputation
|
||||
#' exclusion from FY2002 (`SB087`), the FY2012 dense-to-sparse
|
||||
#' representation change (`SB194`), and the FY2017 government-identifier
|
||||
#' change (`SB086`). These are selected on the break-year window alone.
|
||||
#'
|
||||
#' `SB194` is the one most likely to matter: a query spanning FY2011 to FY2012
|
||||
#' crosses the boundary where an absent cell stops meaning "Census published
|
||||
#' $0" and starts meaning "not reported".
|
||||
#' @section Other provenance blocks:
|
||||
#' `transformations$units_conversion` records the `$1,000s`-to-dollars
|
||||
#' multiply that every amount column has already had applied.
|
||||
#' `transformations$per_capita` records the population denominator and its
|
||||
#' year range. `coverage` and `coverage_mode` appear on multi-government
|
||||
#' results (see [cog_geographic_rollup()]). `completion` appears when
|
||||
#' `complete = TRUE`. `balance_caveats` appears on [cog_balances()] results.
|
||||
#' @export
|
||||
cog_explain <- function(result, format = c("print", "list")) {
|
||||
format <- match.arg(format)
|
||||
@@ -149,15 +125,8 @@ cog_explain <- function(result, format = c("print", "list")) {
|
||||
if (length(prov$suggestions) > 0L) {
|
||||
cli::cli_h2("Suggestions")
|
||||
sugg_lines <- vapply(prov$suggestions, function(s) {
|
||||
line <- sprintf("%s -- %s (years %s-%s): %s", s$recipe_id, s$label,
|
||||
sprintf("%s -- %s (years %s-%s): %s", s$recipe_id, s$label,
|
||||
s$available_years[1], s$available_years[2], s$hint)
|
||||
if (isTRUE(s$suppressed_amount > 0)) {
|
||||
line <- paste0(line, sprintf(" [$%s excluded from %s: %s]",
|
||||
formatC(s$suppressed_amount, format = "f", digits = 0, big.mark = ","),
|
||||
paste0("FY", s$suppressed_years, collapse = ", "),
|
||||
paste(s$suppressed_codes, collapse = ", ")))
|
||||
}
|
||||
line
|
||||
}, character(1))
|
||||
cli::cli_ul(sugg_lines)
|
||||
}
|
||||
|
||||
+1
-1
@@ -28,7 +28,7 @@
|
||||
"*" = "{.code Sys.setenv(USCOGDATA_URL = \"<url-or-local-path>/\")}",
|
||||
"*" = "{.code options(uscogdata.url = \"<url-or-local-path>/\")}",
|
||||
i = "For an offline smoke test, use the bundled fixture: {.code system.file(\"extdata/fixture_corpus\", package = \"uscogdata\")}.",
|
||||
i = "The public corpus is the default: unset USCOGDATA_URL to use it, or point it at a local copy made by {.code cog_mirror()}."
|
||||
i = "For the live Civilytics corpus, request the Nextcloud share URL from the package maintainer."
|
||||
), class = "uscogdata_url_not_configured")
|
||||
}
|
||||
invisible(url)
|
||||
|
||||
@@ -1,81 +0,0 @@
|
||||
# R/pagination.R
|
||||
#
|
||||
# Shared limit/offset machinery. #39 established the semantics inside
|
||||
# .verb_spendrev(); #57 extends them to cog_gov_search() and cog_balances(),
|
||||
# which is what made a single definition worth having: three inline copies of
|
||||
# "coerce, refuse, unwrap the count" would be three places for the meaning of
|
||||
# `total_rows` to drift.
|
||||
#
|
||||
# The SQL side stays in .build_verb_sql() (R/spending.R) -- it already wraps
|
||||
# the aggregate in an outer SELECT so COUNT(*) OVER() sees the post-GROUP-BY
|
||||
# row count rather than the pre-aggregation one, and that is the subtle part
|
||||
# worth not duplicating either.
|
||||
|
||||
#' Coerce and check a limit/offset pair.
|
||||
#'
|
||||
#' Returns the coerced pair, or NULL for `limit` when no page was requested.
|
||||
#' `offset` defaults to 0 whenever `limit` is set, so a caller can supply just
|
||||
#' `limit` and get the first page.
|
||||
#'
|
||||
#' Conflicts with other arguments are deliberately NOT checked here: they
|
||||
#' differ per verb (`complete`/`recipe` for the money verbs, basket mode for
|
||||
#' `cog_gov_search()`, `recipe` alone for `cog_balances()`), and a shared
|
||||
#' function taking a list of conflict flags would be harder to read than the
|
||||
#' three explicit refusals at the call sites.
|
||||
#' @noRd
|
||||
.validate_pagination <- function(limit, offset) {
|
||||
if (is.null(limit)) {
|
||||
return(list(limit = NULL, offset = NULL))
|
||||
}
|
||||
limit <- as.integer(limit)
|
||||
if (length(limit) != 1L || is.na(limit) || limit < 0L) {
|
||||
cli::cli_abort("`limit` must be a single non-negative integer.",
|
||||
class = "uscogdata_invalid_pagination")
|
||||
}
|
||||
offset <- if (is.null(offset)) 0L else as.integer(offset)
|
||||
if (length(offset) != 1L || is.na(offset) || offset < 0L) {
|
||||
cli::cli_abort("`offset` must be a single non-negative integer.",
|
||||
class = "uscogdata_invalid_pagination")
|
||||
}
|
||||
list(limit = limit, offset = offset)
|
||||
}
|
||||
|
||||
#' Wrap a query so one page comes back carrying the unpaginated total.
|
||||
#'
|
||||
#' `COUNT(*) OVER()` rides along as an ordinary column, so the caller gets the
|
||||
#' true total from the SAME scan instead of a second round trip. The outer
|
||||
#' `SELECT *` matters: appending LIMIT/OFFSET directly to a grouped query would
|
||||
#' have the window function count pre-aggregation rows.
|
||||
#' @noRd
|
||||
.paginate_sql <- function(base_sql, limit, offset) {
|
||||
if (is.null(limit)) return(base_sql)
|
||||
sprintf(
|
||||
"SELECT *, COUNT(*) OVER() AS pagination_total_rows
|
||||
FROM (%s) AS _paged
|
||||
LIMIT %d OFFSET %d",
|
||||
base_sql, limit, offset
|
||||
)
|
||||
}
|
||||
|
||||
#' Strip the count column back out and report the unpaginated total.
|
||||
#'
|
||||
#' Returns `list(result = , total_rows = )`.
|
||||
#'
|
||||
#' An empty page -- an offset past the end -- carries no row to read the window
|
||||
#' function off, so that one case falls back to a second, unpaginated
|
||||
#' `COUNT(*)` rather than reporting a wrong zero. `unpaged_sql` is passed as a
|
||||
#' function so the fallback query is only BUILT when it is actually needed;
|
||||
#' every caller's unpaginated SQL is otherwise constructed on every paged call
|
||||
#' and thrown away.
|
||||
#' @noRd
|
||||
.take_pagination_total <- function(result, con, unpaged_sql) {
|
||||
if (nrow(result) > 0L) {
|
||||
total <- result$pagination_total_rows[[1]]
|
||||
result$pagination_total_rows <- NULL
|
||||
return(list(result = result, total_rows = as.integer(total)))
|
||||
}
|
||||
count_sql <- sprintf("SELECT COUNT(*) AS n FROM (%s) AS _uncounted",
|
||||
if (is.function(unpaged_sql)) unpaged_sql() else unpaged_sql)
|
||||
list(result = result,
|
||||
total_rows = as.integer(DBI::dbGetQuery(con, count_sql)$n[[1]]))
|
||||
}
|
||||
@@ -240,23 +240,6 @@ cog_find_peers <- function(target_govid,
|
||||
#' group_by(year) |>
|
||||
#' summarise(p50 = quantile(total, 0.5, na.rm = TRUE))
|
||||
#' ```
|
||||
#' @section Reading `coverage`:
|
||||
#' `provenance$coverage` reports `n_units_reporting` against
|
||||
#' `n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
||||
#' it counts cohort members with rows for the category you asked for, not
|
||||
#' cohort members collected that year.** A government that was surveyed and
|
||||
#' genuinely spends nothing in that category is indistinguishable here from one
|
||||
#' that was never surveyed.
|
||||
#'
|
||||
#' The ratio is therefore **not a response rate** and must not be used as one.
|
||||
#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||
#' `category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||
#' contract policing to the county sheriff, not non-response.
|
||||
#'
|
||||
#' The comparison that *is* valid is the same category across a census year
|
||||
#' (ending in 2 or 7) and a sample year, where the real-zero component is
|
||||
#' roughly constant and the difference reflects the survey cycle. `is_census_year`
|
||||
#' marks which is which.
|
||||
#' @export
|
||||
cog_peer_compare <- function(target_govid, peers, category, years,
|
||||
per_capita = TRUE, adjust_to_year = NULL,
|
||||
|
||||
+1
-9
@@ -67,15 +67,7 @@
|
||||
basis_note = basis_note,
|
||||
expenditure_concept = expenditure_concept,
|
||||
expenditure_concept_note = expenditure_concept_note,
|
||||
# isTRUE() alone would collapse a deliberate NA (all-categories mode,
|
||||
# where suppression detection cannot run -- see .verb_spendrev()) down to
|
||||
# FALSE, turning "we don't know" back into the false claim this field
|
||||
# exists to avoid. Preserve NA; otherwise normalize to a strict logical.
|
||||
expenditure_concept_direct_suppressed = if (isTRUE(is.na(expenditure_concept_direct_suppressed))) {
|
||||
NA
|
||||
} else {
|
||||
isTRUE(expenditure_concept_direct_suppressed)
|
||||
},
|
||||
expenditure_concept_direct_suppressed = isTRUE(expenditure_concept_direct_suppressed),
|
||||
revenue_concept = revenue_concept,
|
||||
harmonization = harmonization %||% list(
|
||||
applied = FALSE, na_rows_excluded = 0L, na_amount_excluded = 0,
|
||||
|
||||
+3
-3
@@ -113,7 +113,7 @@ cog_recipes <- function(pattern = NULL) {
|
||||
#' `year_min`/`year_max` -- so there is no double-counting. (Checkpoint
|
||||
#' review docs/phase_r_harmonization_review.md § 0.2.)
|
||||
#' @noRd
|
||||
.run_recipe <- function(con, recipe_id, cohort, years) {
|
||||
.run_recipe <- function(con, recipe_id, govid, years) {
|
||||
sql <- sprintf(
|
||||
"SELECT l.year, l.canonical_govid,
|
||||
COALESCE(x.gov_name, l.gov_name) AS gov_name,
|
||||
@@ -128,11 +128,11 @@ cog_recipes <- function(pattern = NULL) {
|
||||
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
|
||||
WHERE r.recipe_id = %1$s
|
||||
AND %2$s
|
||||
AND l.canonical_govid IN (%2$s)
|
||||
AND l.year IN (%3$s)
|
||||
GROUP BY 1, 2, 3
|
||||
ORDER BY 1, 2",
|
||||
.sql_lit_chr(recipe_id), .cohort_sql(cohort, "l.canonical_govid"),
|
||||
.sql_lit_chr(recipe_id), .sql_lit_chr(govid),
|
||||
paste(as.integer(years), collapse = ",")
|
||||
)
|
||||
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
|
||||
+3
-19
@@ -8,17 +8,6 @@
|
||||
#' multiplies by 1000 and records the conversion in `provenance`).
|
||||
#'
|
||||
#' @inheritParams cog_spending
|
||||
#' @param category Character vector of category names (from
|
||||
#' `summary_categories.category`), or `NULL` for all categories broken out
|
||||
#' one row each. The reserved value `"All Categories"` instead returns a
|
||||
#' single summed row per `(year, canonical_govid, subtype)`, covering every
|
||||
#' category inside the requested concept's subtype scope. It cannot be
|
||||
#' combined with other category names, and it is not the same thing as
|
||||
#' `revenue_concept = "total"`: the concept chooses which subtypes are in
|
||||
#' scope, `"All Categories"` chooses whether rows inside that scope are
|
||||
#' broken out or summed. Because the result keeps one row per
|
||||
#' `revenue_subtype`, filtering the returned frame to
|
||||
#' `revenue_subtype == "own_source"` gives an own-source revenue total.
|
||||
#' @param revenue_concept Which of Census's two published revenue concepts to
|
||||
#' return. Concepts are defined as sets of the crosswalk's `revenue_subtype`
|
||||
#' values -- never as item-code first letters, which cannot classify
|
||||
@@ -50,12 +39,11 @@
|
||||
#' optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`,
|
||||
#' and `value_source` when `complete = TRUE`.
|
||||
#' @export
|
||||
cog_revenue <- function(govid = NULL, years, category = NULL,
|
||||
cog_revenue <- function(govid, years, category = NULL,
|
||||
per_capita = FALSE, adjust_to_year = NULL,
|
||||
basis = c("harmonized", "raw"), recipe = NULL,
|
||||
revenue_concept = c("general", "total"),
|
||||
complete = FALSE, limit = NULL, offset = NULL,
|
||||
state = NULL, type = NULL) {
|
||||
complete = FALSE) {
|
||||
# flow_prefixes no longer classifies rows (crosswalk revenue_subtype
|
||||
# membership does -- General Revenue, i.e. everything except
|
||||
# insurance_trust) -- it only scopes the recipe-suggestion machinery to
|
||||
@@ -74,10 +62,6 @@ cog_revenue <- function(govid = NULL, years, category = NULL,
|
||||
basis = basis,
|
||||
recipe = recipe,
|
||||
revenue_concept = revenue_concept,
|
||||
complete = complete,
|
||||
limit = limit,
|
||||
offset = offset,
|
||||
state = state,
|
||||
type = type
|
||||
complete = complete
|
||||
)
|
||||
}
|
||||
|
||||
+1
-22
@@ -19,11 +19,7 @@
|
||||
#' `state`, `county`, `city`. Each element is a character vector of
|
||||
#' `canonical_govid` values. At least one layer required.
|
||||
#' @param category Single category name or character vector (passed through
|
||||
#' to [cog_spending()]), or the reserved `"All Categories"` for one summed
|
||||
#' row per `(year, canonical_govid, subtype)` covering every category in the
|
||||
#' concept's scope. `"All Categories"` is the efficient way to build a
|
||||
#' geographic total: without it a caller must issue one rollup per category
|
||||
#' and sum the results themselves.
|
||||
#' to [cog_spending()]).
|
||||
#' @param years Integer vector of years.
|
||||
#' @param per_capita If `TRUE`, per-capita uses each gov's own per-year
|
||||
#' population from `gov_population_yearly`. Govs with missing population
|
||||
@@ -60,23 +56,6 @@
|
||||
#' `codes_included`, `aggregate_fallback`, `scope_note`, `notes`. Carries a
|
||||
#' `provenance` attribute with `verb = "cog_geographic_rollup"`, `layers`,
|
||||
#' and `rollup$included_govids` / `rollup$excluded_govids`.
|
||||
#' @section Reading `coverage`:
|
||||
#' `provenance$coverage` reports `n_units_reporting` against
|
||||
#' `n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
||||
#' it counts governments with rows for the category you asked for, not
|
||||
#' governments collected that year.** A government that was surveyed and
|
||||
#' genuinely spends nothing in that category is indistinguishable here from one
|
||||
#' that was never surveyed.
|
||||
#'
|
||||
#' The ratio is therefore **not a response rate** and must not be used as one.
|
||||
#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||
#' `category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||
#' contract policing to the county sheriff, not non-response.
|
||||
#'
|
||||
#' The comparison that *is* valid is the same category across a census year
|
||||
#' (ending in 2 or 7) and a sample year, where the real-zero component is
|
||||
#' roughly constant and the difference reflects the survey cycle. `is_census_year`
|
||||
#' marks which is which.
|
||||
#' @export
|
||||
cog_geographic_rollup <- function(govids, category, years,
|
||||
per_capita = FALSE, adjust_to_year = NULL,
|
||||
|
||||
+11
-61
@@ -44,22 +44,10 @@
|
||||
#' in basket mode (recycles from length 1). Excluded types `4`/`5` (or
|
||||
#' `"special_district"` / `"school_district"`) trigger an explanatory
|
||||
#' message and an empty result.
|
||||
#' @param limit Maximum number of rows to return, applied in SQL. `NULL`
|
||||
#' (default) returns every match -- which, with no other filter, is the
|
||||
#' entire crosswalk. Utility mode only: pagination has no meaning in basket
|
||||
#' mode, where the result is one resolved row per requested name in input
|
||||
#' order, and is refused there with class
|
||||
#' `uscogdata_basket_pagination_conflict`.
|
||||
#' @param offset Rows to skip before `limit` starts counting (0-based).
|
||||
#' Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set.
|
||||
#' @return A tibble of `canonical_fips_xwalk` rows. In utility mode, all
|
||||
#' matches sorted by `population_acs` desc, ties broken by
|
||||
#' `canonical_govid`. In basket mode, resolved rows in input order, with
|
||||
#' `attr(., "resolution")` set to the sidecar tibble.
|
||||
#'
|
||||
#' When `limit` is set, carries a `total_rows` attribute: the full
|
||||
#' unpaginated match count, computed by the same query (`COUNT(*) OVER()`)
|
||||
#' rather than a second scan.
|
||||
#' matches sorted by `population_acs` desc. In basket mode, resolved
|
||||
#' rows in input order, with `attr(., "resolution")` set to the
|
||||
#' sidecar tibble.
|
||||
#' @seealso [cog_basket_resolution()], [cog_basket_unresolved()],
|
||||
#' [cog_spending()], [cog_revenue()].
|
||||
#' @examples
|
||||
@@ -94,12 +82,7 @@
|
||||
#' )
|
||||
#' }
|
||||
#' @export
|
||||
cog_gov_search <- function(name = NULL, state = NULL, type = NULL,
|
||||
limit = NULL, offset = NULL) {
|
||||
paging <- .validate_pagination(limit, offset)
|
||||
limit <- paging$limit
|
||||
offset <- paging$offset
|
||||
|
||||
cog_gov_search <- function(name = NULL, state = NULL, type = NULL) {
|
||||
if (!is.null(type) && length(type) == 1L && .is_excluded_type(type)) {
|
||||
cli::cli_inform(c(
|
||||
i = "v0.1 covers gov_types 0-3 (state/county/city/township) only.",
|
||||
@@ -111,18 +94,6 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL,
|
||||
con <- .ensure_session()
|
||||
|
||||
if (length(name) > 1L) {
|
||||
# Basket mode returns one resolved row per requested name, in input order,
|
||||
# with a resolution sidecar describing how each was matched. A page of that
|
||||
# is not a page of anything the caller asked for -- the sidecar would still
|
||||
# describe every name -- so refuse rather than silently ignoring the
|
||||
# arguments. Same shape as the recipe/complete refusals in .verb_spendrev().
|
||||
if (!is.null(limit)) {
|
||||
cli::cli_abort(c(
|
||||
"`limit`/`offset` cannot be combined with basket mode.",
|
||||
"i" = "Basket mode ({.code length(name) > 1}) returns one resolved row per requested name, in input order, with a resolution sidecar covering all of them.",
|
||||
"*" = "Drop `limit`/`offset`, or search one name at a time."
|
||||
), class = "uscogdata_basket_pagination_conflict")
|
||||
}
|
||||
return(.resolve_basket(name = name, state = state, type = type, con = con))
|
||||
}
|
||||
|
||||
@@ -152,26 +123,12 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL,
|
||||
}
|
||||
|
||||
where <- if (length(preds) == 0L) "" else paste("WHERE", paste(preds, collapse = " AND "))
|
||||
# canonical_govid breaks ties. population_acs alone is NOT a total order --
|
||||
# governments sharing a population, and the whole NULLS LAST block, came back
|
||||
# in whatever order the scan produced. That was invisible while every call
|
||||
# returned the full result set, but it makes a paged sweep unsound: two
|
||||
# requests can order the tied rows differently, so a row is duplicated on one
|
||||
# page and missing from the next. Any pagination has to sit on a total order.
|
||||
base_sql <- paste(
|
||||
sql <- paste(
|
||||
"SELECT * FROM canonical_fips_xwalk",
|
||||
where,
|
||||
"ORDER BY population_acs DESC NULLS LAST, canonical_govid"
|
||||
"ORDER BY population_acs DESC NULLS LAST"
|
||||
)
|
||||
result <- tibble::as_tibble(
|
||||
DBI::dbGetQuery(con, .paginate_sql(base_sql, limit, offset))
|
||||
)
|
||||
if (is.null(limit)) return(result)
|
||||
|
||||
paged <- .take_pagination_total(result, con, base_sql)
|
||||
out <- paged$result
|
||||
attr(out, "total_rows") <- paged$total_rows
|
||||
out
|
||||
tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
}
|
||||
|
||||
#' @noRd
|
||||
@@ -227,18 +184,11 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL,
|
||||
if (!is.character(state) || length(state) != 1L) {
|
||||
cli::cli_abort("`state` must be a 2-letter USPS abbrev or a FIPS integer.")
|
||||
}
|
||||
# Membership tested before the lookup, not after: `.state_abbrev_to_fips` is
|
||||
# a named CHARACTER vector, and `[[` on a name it does not carry throws
|
||||
# base R's "subscript out of bounds" rather than returning NULL -- which
|
||||
# made the curated message below unreachable dead code. Reported as a bare
|
||||
# subscript error, `cog_gov_search(state = "ZZ")` gave no hint that the
|
||||
# argument wants a postal abbreviation.
|
||||
key <- toupper(state)
|
||||
if (!key %in% names(.state_abbrev_to_fips)) {
|
||||
cli::cli_abort("Unknown state abbreviation: {state}.",
|
||||
class = "uscogdata_unknown_state")
|
||||
fips <- .state_abbrev_to_fips[[toupper(state)]]
|
||||
if (is.null(fips)) {
|
||||
cli::cli_abort("Unknown state abbreviation: {state}.")
|
||||
}
|
||||
.state_abbrev_to_fips[[key]]
|
||||
fips
|
||||
}
|
||||
|
||||
# USPS state / territory abbreviation -> 2-digit FIPS code.
|
||||
|
||||
+1
-31
@@ -2,23 +2,13 @@
|
||||
|
||||
#' Internal: open session, register views, cache manifest.
|
||||
#' Not exported. Called lazily by verbs via .ensure_session().
|
||||
#'
|
||||
#' `threads` and `memory_limit` default to the resolved configuration and are
|
||||
#' applied as pragmas on the new connection. When both resolve to NULL -- which
|
||||
#' is the case unless the operator sets one -- NO pragma is issued at all, so an
|
||||
#' unconfigured session connects exactly as it did before this argument existed.
|
||||
#' @noRd
|
||||
cog_open <- function(url = .resolve_url(),
|
||||
cache_dir = .resolve_cache_dir(),
|
||||
threads = .resolve_duckdb_threads(),
|
||||
memory_limit = .resolve_duckdb_memory_limit()) {
|
||||
cache_dir = .resolve_cache_dir()) {
|
||||
.check_url_configured(url)
|
||||
if (!dir.exists(cache_dir)) dir.create(cache_dir, recursive = TRUE)
|
||||
|
||||
con <- DBI::dbConnect(duckdb::duckdb())
|
||||
# Before anything else touches the connection: httpfs reads the corpus, and
|
||||
# a remote read should already be bound by whatever budget the operator set.
|
||||
.apply_duckdb_limits(con, threads, memory_limit)
|
||||
DBI::dbExecute(con, "INSTALL httpfs; LOAD httpfs;")
|
||||
|
||||
manifest <- .fetch_or_cache_manifest(url, cache_dir)
|
||||
@@ -35,26 +25,6 @@ cog_open <- function(url = .resolve_url(),
|
||||
invisible(con)
|
||||
}
|
||||
|
||||
#' Apply the operator's DuckDB resource budget to a fresh connection.
|
||||
#'
|
||||
#' Split out from cog_open() so the "unset changes nothing" property is one
|
||||
#' readable branch rather than two conditionals buried in the connection path.
|
||||
#' Both settings are session-scoped in DuckDB, so this must run per connection;
|
||||
#' cog_close() discards the connection and the next cog_open() re-resolves,
|
||||
#' which is what makes a changed option take effect on the next session.
|
||||
#' @noRd
|
||||
.apply_duckdb_limits <- function(con, threads, memory_limit) {
|
||||
if (!is.null(threads)) {
|
||||
DBI::dbExecute(con, sprintf("SET threads TO %d", threads))
|
||||
}
|
||||
if (!is.null(memory_limit)) {
|
||||
# Quoted as a string literal: DuckDB's memory_limit takes '4GB', not 4GB.
|
||||
DBI::dbExecute(con, sprintf("SET memory_limit TO %s",
|
||||
.sql_lit_chr(memory_limit)))
|
||||
}
|
||||
invisible(con)
|
||||
}
|
||||
|
||||
#' @noRd
|
||||
.ensure_session <- function() {
|
||||
if (is.null(.uscogdata_env$con) ||
|
||||
|
||||
+35
-284
@@ -19,14 +19,6 @@
|
||||
.spend_subtypes_primary <- c("operations", "capital", "assistance")
|
||||
.spend_subtypes_direct <- c(.spend_subtypes_primary, "interest", "insurance_benefits")
|
||||
|
||||
# The reserved pseudo-category. Deliberately NOT "Total": `category = "Total"`
|
||||
# would sit one argument away from `expenditure_concept = "total"` and mean
|
||||
# something different -- the concept selects WHICH SUBTYPES are in scope, this
|
||||
# selects whether the rows inside that scope are broken out by category or
|
||||
# summed. "All Categories" states the operation and cannot be misread as the
|
||||
# concept.
|
||||
.ALL_CATEGORIES <- "All Categories"
|
||||
|
||||
#' @noRd
|
||||
.expenditure_concept_subtypes <- function(concept) {
|
||||
switch(concept,
|
||||
@@ -68,21 +60,10 @@
|
||||
#' millions/billions). The conversion is recorded in the provenance attribute
|
||||
#' under `transformations$units_conversion`.
|
||||
#'
|
||||
#' @param govid Character vector of `canonical_govid` values, or `NULL` to name
|
||||
#' the cohort by `state`/`type` instead. One of `govid`, `state`, or `type`
|
||||
#' is required.
|
||||
#' @param govid Character vector of `canonical_govid` values.
|
||||
#' @param years Integer vector of years.
|
||||
#' @param category Character vector of category names (from
|
||||
#' `summary_categories.category`), or `NULL` for all categories broken out
|
||||
#' one row each. The reserved value `"All Categories"` instead returns a
|
||||
#' single summed row per `(year, canonical_govid, subtype)`, covering every
|
||||
#' category inside the requested concept's subtype scope. It cannot be
|
||||
#' combined with other category names, and it is not the same thing as
|
||||
#' `expenditure_concept = "total"`: the concept chooses which subtypes are in
|
||||
#' scope, `"All Categories"` chooses whether rows inside that scope are
|
||||
#' broken out or summed. Because the result keeps one row per
|
||||
#' `spend_subtype`, filtering the returned frame to
|
||||
#' `spend_subtype == "operations"` gives an operating-expenditure total.
|
||||
#' `summary_categories.category`), or `NULL` for all categories.
|
||||
#' @param per_capita If `TRUE`, adds `amt_per_capita_nominal` (and
|
||||
#' `amt_per_capita_real` when `adjust_to_year` is set) using the per-year
|
||||
#' Census F-33 population from `gov_population_yearly`. Result also gains
|
||||
@@ -152,12 +133,7 @@
|
||||
#' component (when one exists), and
|
||||
#' `provenance$expenditure_concept_direct_suppressed` is `TRUE` -- the
|
||||
#' figure in those rows is the intergovernmental leg alone, not Direct +
|
||||
#' IG. When `category = "All Categories"` is combined with
|
||||
#' `expenditure_concept = "total"`, this detection cannot run (it keys on
|
||||
#' per-category rows, which all-categories mode collapses to one literal
|
||||
#' value), so `expenditure_concept_direct_suppressed` is `NA` rather than a
|
||||
#' possibly-false `FALSE`; query an explicit `category` to get a real
|
||||
#' answer.
|
||||
#' IG.
|
||||
#' @param complete If `TRUE`, fill the requested grid so that a cell the
|
||||
#' corpus does not carry still appears, labelled with **why** it is
|
||||
#' missing, and add a `value_source` column to every row:
|
||||
@@ -180,36 +156,6 @@
|
||||
#' `recipe` or with `expenditure_concept = "total"` (class
|
||||
#' `uscogdata_complete_unsupported`) — neither draws its cells from
|
||||
#' `code_set`.
|
||||
#' @param limit Maximum number of result rows to return, pushed into the SQL
|
||||
#' query itself (`LIMIT`/`OFFSET`) rather than applied after the full
|
||||
#' result is materialized. `NULL` (the default) returns every matching row,
|
||||
#' exactly as before this parameter existed. Mutually exclusive with
|
||||
#' `recipe` and with `complete = TRUE` -- see `offset` and `total_rows`.
|
||||
#' @param offset Rows to skip before `limit` starts counting (0-based).
|
||||
#' Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set.
|
||||
#' @param state,type Name the cohort by predicate instead of by id: `state` is
|
||||
#' a 2-letter USPS abbreviation (or a FIPS code) and `type` is one of
|
||||
#' `"state"`, `"county"`, `"city"`, `"township"` (or the integer `0:3`) --
|
||||
#' the same vocabulary, and the same internal coercion, as
|
||||
#' [cog_gov_search()]. Both default to `NULL`.
|
||||
#'
|
||||
#' The cohort is then expressed as a subquery against `canonical_fips_xwalk`
|
||||
#' inside each statement rather than round-tripped through R as a literal id
|
||||
#' list. For a fleet-scale cohort that is the difference between a
|
||||
#' 301,591-character `IN` list re-parsed in 5--8 statements per call and a
|
||||
#' constant-size predicate: measured at **94 ms versus 449 ms** for the same
|
||||
#' FY2022 aggregate over the 20,106-government `type = "city"` cohort, within
|
||||
#' 7% of the no-filter floor.
|
||||
#'
|
||||
#' Supplying `govid` **and** `state`/`type` INTERSECTS them -- the
|
||||
#' governments in `govid` that also match the predicate -- rather than one
|
||||
#' silently taking precedence. Naming no cohort at all (`govid`, `state` and
|
||||
#' `type` all `NULL`) aborts with class `uscogdata_no_cohort`.
|
||||
#'
|
||||
#' When the cohort is named by predicate, `provenance$scope$govids_found`
|
||||
#' and `govids_missing` are empty -- there is no id list to report against --
|
||||
#' and `provenance$scope$cohort` carries `state`, `type` and
|
||||
#' `n_governments` instead. A `govid`-named cohort reports exactly as before.
|
||||
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||
#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
||||
#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
||||
@@ -217,18 +163,13 @@
|
||||
#' and `value_source` when `complete = TRUE`.
|
||||
#' Carries a `provenance` attribute matching `inst/schemas/provenance-v1.json`,
|
||||
#' whose `completion` block reports `applied`, `rows_filled`, and the
|
||||
#' per-year `absence_means` rule that was applied. When `limit` is set,
|
||||
#' also carries a `total_rows` attribute: the full unpaginated row count,
|
||||
#' computed by the same query (`COUNT(*) OVER()`) rather than a second
|
||||
#' round trip -- so a caller walking pages never has to ask "how many are
|
||||
#' there" separately.
|
||||
#' per-year `absence_means` rule that was applied.
|
||||
#' @export
|
||||
cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
cog_spending <- function(govid, years, category = NULL,
|
||||
per_capita = FALSE, adjust_to_year = NULL,
|
||||
basis = c("harmonized", "raw"), recipe = NULL,
|
||||
expenditure_concept = c("primary", "direct", "total"),
|
||||
complete = FALSE, limit = NULL, offset = NULL,
|
||||
state = NULL, type = NULL) {
|
||||
complete = FALSE) {
|
||||
# flow_prefixes no longer classifies rows (crosswalk subtype membership
|
||||
# does, per expenditure_concept) -- it only scopes the recipe-suggestion
|
||||
# machinery to this verb's recipe families (see R/suggestions.R; the
|
||||
@@ -247,11 +188,7 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
basis = basis,
|
||||
recipe = recipe,
|
||||
expenditure_concept = expenditure_concept,
|
||||
complete = complete,
|
||||
limit = limit,
|
||||
offset = offset,
|
||||
state = state,
|
||||
type = type
|
||||
complete = complete
|
||||
)
|
||||
}
|
||||
|
||||
@@ -279,8 +216,7 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
basis = c("harmonized", "raw"), recipe = NULL,
|
||||
expenditure_concept = c("primary", "direct", "total"),
|
||||
revenue_concept = c("general", "total"),
|
||||
complete = FALSE, limit = NULL, offset = NULL,
|
||||
state = NULL, type = NULL) {
|
||||
complete = FALSE) {
|
||||
basis_explicit <- length(basis) == 1L
|
||||
basis <- match.arg(basis, c("harmonized", "raw"))
|
||||
# match.arg() itself throws a base `simpleError`, not an rlang-classed
|
||||
@@ -320,32 +256,9 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
.revenue_concept_subtypes(revenue_concept)
|
||||
}
|
||||
|
||||
# NULL `govid` means "the cohort is named by predicate"; anything else is
|
||||
# coerced and validated exactly as before, so an empty or wrong-typed vector
|
||||
# still fails with its original message rather than being read as absent.
|
||||
govid <- if (is.null(govid)) NULL else .coerce_govid_input(govid, arg = "govid")
|
||||
# allow_all_categories = TRUE: cog_spending()/cog_revenue() are the two
|
||||
# verbs the reserved pseudo-category is defined for. cog_balances() shares
|
||||
# this validator but leaves the argument at its FALSE default, so it
|
||||
# rejects "All Categories" instead of silently returning zero rows
|
||||
# (finding 3, all-categories review).
|
||||
govid <- .coerce_govid_input(govid, arg = "govid")
|
||||
.validate_verb_inputs(govid, years, category, per_capita, adjust_to_year,
|
||||
recipe, allow_all_categories = TRUE)
|
||||
|
||||
# Built after validation so the argument-shape errors above keep firing
|
||||
# first, and before .ensure_session() so a bad state/type costs no I/O.
|
||||
cohort <- .make_cohort(govid, state, type)
|
||||
|
||||
# Recognize the reserved pseudo-category. Detected after type validation so a
|
||||
# non-character `category` still fails with the ordinary type error.
|
||||
all_categories <- !is.null(category) && .ALL_CATEGORIES %in% category
|
||||
if (all_categories && length(category) > 1L) {
|
||||
cli::cli_abort(c(
|
||||
"{.val {(.ALL_CATEGORIES)}} cannot be combined with other categories.",
|
||||
"i" = "It already sums every category in the requested concept's scope.",
|
||||
"*" = "Ask for it alone, or list the specific categories you want."
|
||||
), class = "uscogdata_all_categories_not_combinable")
|
||||
}
|
||||
recipe)
|
||||
|
||||
if (!is.null(recipe) && identical(expenditure_concept, "total")) {
|
||||
cli::cli_abort(c(
|
||||
@@ -386,37 +299,6 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
"Use `expenditure_concept = \"direct\"` with `complete = TRUE`, or drop `complete`."
|
||||
)
|
||||
}
|
||||
if (complete && all_categories) {
|
||||
.abort_complete_unsupported(
|
||||
"`category = \"All Categories\"` collapses the category dimension that `code_set` grids over (see `.completion_grid_sql()`), so there is no per-category grid left to fill -- filling a summed row has no defined semantics.",
|
||||
"Drop `complete`, or use `complete = TRUE` with an explicit `category` (or `category = NULL` for every category)."
|
||||
)
|
||||
}
|
||||
|
||||
# limit/offset push the page into the SQL itself (see .build_verb_sql()),
|
||||
# so the two things that would make "a page of what" ambiguous are refused
|
||||
# up front rather than silently ignored: complete = TRUE fills a grid over
|
||||
# the FULL requested (year, category) space, and a recipe's result comes
|
||||
# from .run_recipe()'s own query, which this function does not touch.
|
||||
paging <- .validate_pagination(limit, offset)
|
||||
limit <- paging$limit
|
||||
offset <- paging$offset
|
||||
if (!is.null(limit)) {
|
||||
if (complete) {
|
||||
cli::cli_abort(c(
|
||||
"`limit`/`offset` cannot be combined with `complete = TRUE`.",
|
||||
"i" = "`complete` fills a grid over the FULL requested (year, category) space; paginating a slice of already-grouped rows has no defined meaning for the cells it would fill.",
|
||||
"*" = "Drop `limit`/`offset`, or drop `complete`."
|
||||
), class = "uscogdata_complete_pagination_conflict")
|
||||
}
|
||||
if (!is.null(recipe)) {
|
||||
cli::cli_abort(c(
|
||||
"`limit`/`offset` cannot be combined with `recipe`.",
|
||||
"i" = "A recipe's result comes from a separate query (`.run_recipe()`) that pagination is not wired into yet.",
|
||||
"*" = "Drop `limit`/`offset`, or drop `recipe`."
|
||||
), class = "uscogdata_recipe_pagination_conflict")
|
||||
}
|
||||
}
|
||||
|
||||
years <- as.integer(years)
|
||||
if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year)
|
||||
@@ -430,13 +312,12 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
|
||||
recipe_block <- NULL
|
||||
category_for_prov <- category
|
||||
total_rows <- NULL # set below only when limit is non-NULL (non-recipe path)
|
||||
if (!is.null(recipe)) {
|
||||
.require_schema_v5(con, manifest, "recipe =")
|
||||
.validate_recipe_id(con, recipe)
|
||||
comps <- .recipe_components(con, recipe)
|
||||
recipe_label <- comps$label[[1]]
|
||||
result <- .run_recipe(con, recipe, cohort, years)
|
||||
result <- .run_recipe(con, recipe, govid, years)
|
||||
sql <- attr(result, "sql_query")
|
||||
result <- .shape_recipe_result(result, subtype_col, recipe_label)
|
||||
recipe_block <- list(
|
||||
@@ -452,22 +333,9 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
} else {
|
||||
NULL
|
||||
}
|
||||
sql <- .build_verb_sql(view, subtype_col, cohort, years,
|
||||
if (all_categories) NULL else category,
|
||||
ig_view, subtype_scope,
|
||||
all_categories = all_categories,
|
||||
limit = limit, offset = offset)
|
||||
sql <- .build_verb_sql(view, subtype_col, govid, years, category, ig_view,
|
||||
subtype_scope)
|
||||
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
if (!is.null(limit)) {
|
||||
paged <- .take_pagination_total(result, con, function() {
|
||||
.build_verb_sql(view, subtype_col, cohort, years,
|
||||
if (all_categories) NULL else category,
|
||||
ig_view, subtype_scope,
|
||||
all_categories = all_categories)
|
||||
})
|
||||
result <- paged$result
|
||||
total_rows <- paged$total_rows
|
||||
}
|
||||
}
|
||||
|
||||
# Fill BEFORE per_capita / inflation so the added cells get the same
|
||||
@@ -476,13 +344,13 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
# spurious 0.
|
||||
completion <- list(applied = FALSE, rows_filled = 0L, absence_means = list())
|
||||
if (complete) {
|
||||
result <- .complete_result(result, con, subtype_col, cohort, years,
|
||||
result <- .complete_result(result, con, subtype_col, govid, years,
|
||||
category, subtype_scope)
|
||||
completion <- attr(result, ".completion")
|
||||
attr(result, ".completion") <- NULL
|
||||
}
|
||||
|
||||
if (per_capita) result <- .attach_per_capita(result, con)
|
||||
if (per_capita) result <- .attach_per_capita(result, con, govid)
|
||||
if (!is.null(adjust_to_year)) {
|
||||
result <- .attach_real_dollars(result, adjust_to_year, per_capita)
|
||||
}
|
||||
@@ -510,7 +378,7 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
basis_for_prov <- resolved$basis
|
||||
basis_note_for_prov <- resolved$note
|
||||
harmonization <- .build_harmonization_block(
|
||||
con, cohort, years, resolved, subtype_col, subtype_scope
|
||||
con, govid, years, resolved, subtype_col, subtype_scope
|
||||
)
|
||||
# C1(a): gap detection must run against the Direct leg alone. `result`
|
||||
# can also carry UNION'd intergovernmental rows (expenditure_concept =
|
||||
@@ -525,13 +393,9 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
} else {
|
||||
result
|
||||
}
|
||||
suggestions <- .build_suggestions(con, cohort, years, category,
|
||||
suggestions <- .build_suggestions(con, govid, years, category,
|
||||
direct_leg_result,
|
||||
resolved$basis, flow_prefixes,
|
||||
.select_long_view(view_base, resolved$basis),
|
||||
all_categories = all_categories,
|
||||
subtype_col = subtype_col,
|
||||
subtype_scope = subtype_scope)
|
||||
resolved$basis, flow_prefixes)
|
||||
}
|
||||
|
||||
# C1(b): when expenditure_concept = "total", flag any row where the IG
|
||||
@@ -543,32 +407,13 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
# direct spending in that category, which is correct, ordinary data). When
|
||||
# a covering recipe is found, both the row-level notes and the provenance
|
||||
# say so rather than pass silently as a plausible Total.
|
||||
#
|
||||
# In all-categories mode this cannot run at all: .detect_direct_suppressed()
|
||||
# keys on (year, canonical_govid, category), and every row shares the same
|
||||
# literal "All Categories" value, so the key collides across every real
|
||||
# category for that (year, govid) -- an IG-only row for a suppressed
|
||||
# category becomes indistinguishable from one sharing a key with an
|
||||
# unrelated category's ordinary Direct row. `has_direct` would then read
|
||||
# TRUE whenever the government has ANY direct spending at all, and the
|
||||
# detector could never fire. Rather than run it and report a false FALSE,
|
||||
# skip it and record NA -- the provenance must stop making a claim it
|
||||
# cannot support (finding 1, all-categories review).
|
||||
suppression_unavailable <- all_categories &&
|
||||
identical(expenditure_concept, "total")
|
||||
direct_suppressed_info <- if (suppression_unavailable) {
|
||||
list(flag = rep(NA, nrow(result)), notes = rep(NA_character_, nrow(result)))
|
||||
} else if (identical(expenditure_concept, "total")) {
|
||||
direct_suppressed_info <- if (identical(expenditure_concept, "total")) {
|
||||
.detect_direct_suppressed(con, result, subtype_col)
|
||||
} else {
|
||||
list(flag = rep(FALSE, nrow(result)), notes = rep(NA_character_, nrow(result)))
|
||||
}
|
||||
direct_suppressed <- direct_suppressed_info$flag
|
||||
direct_suppressed_flag <- if (suppression_unavailable) {
|
||||
NA
|
||||
} else {
|
||||
isTRUE(any(direct_suppressed))
|
||||
}
|
||||
direct_suppressed_flag <- isTRUE(any(direct_suppressed))
|
||||
|
||||
result$notes <- .notes_column(result, direct_suppressed_info$notes)
|
||||
|
||||
@@ -577,20 +422,9 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
# leg is suppressed for at least one requested (year, category), append an
|
||||
# explicit warning rather than let the base note's "Total = Direct + IG"
|
||||
# framing stand unqualified for rows where that arithmetic didn't happen.
|
||||
# When suppression detection itself is unavailable (all-categories mode),
|
||||
# say so instead of silently reusing the unqualified base note.
|
||||
expenditure_concept_note_for_prov <- if (identical(expenditure_concept, "total")) {
|
||||
base_note <- "Total = Direct + intergovernmental (M to local govts + L to state govts). Legacy-era IG is assembled from aggregate-flagged rows, which are year-disjoint from their modern leaf components; the L-- family total is excluded."
|
||||
if (suppression_unavailable) {
|
||||
paste0(
|
||||
base_note,
|
||||
" NOTE: direct-leg-suppression detection is unavailable when ",
|
||||
"`category = \"All Categories\"` -- it keys on per-category rows, ",
|
||||
"which this mode collapses. `expenditure_concept_direct_suppressed` ",
|
||||
"is NA here rather than a possibly-false FALSE; query an explicit ",
|
||||
"`category` (or `category = NULL`) to get a real answer."
|
||||
)
|
||||
} else if (isTRUE(direct_suppressed_flag)) {
|
||||
if (direct_suppressed_flag) {
|
||||
paste0(
|
||||
base_note,
|
||||
" NOTE: for at least one requested (year, category) the Direct leg ",
|
||||
@@ -630,45 +464,18 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
)
|
||||
prov$scope$govids_found <- scope$found
|
||||
prov$scope$govids_missing <- scope$missing
|
||||
# A predicate-named cohort has no id list to report found/missing against
|
||||
# (both stay empty), so it describes itself instead. Deliberately a COUNT
|
||||
# rather than the resolved ids: enumerating them would put 20,000 govids in
|
||||
# every fleet-scale response body, which is the cost this path exists to
|
||||
# remove. NULL for a govid-named cohort, so that output is untouched.
|
||||
prov$scope$cohort <- .cohort_provenance(con, cohort)
|
||||
attr(result, "provenance") <- prov
|
||||
attr(result, ".popyear_range") <- NULL
|
||||
# Attached here, after every downstream transform (per_capita/real-dollar
|
||||
# joins, notes, subtype filtering), the same way provenance is -- an
|
||||
# attribute set before those runs is not guaranteed to survive them.
|
||||
if (!is.null(limit)) attr(result, "total_rows") <- total_rows
|
||||
|
||||
if (length(suggestions) > 0L) .inform_suggestions(suggestions)
|
||||
|
||||
result
|
||||
}
|
||||
|
||||
#' Shared input validation for the money/holdings verbs.
|
||||
#'
|
||||
#' `allow_all_categories` gates the reserved pseudo-category
|
||||
#' `.ALL_CATEGORIES` ("All Categories"). It is meaningful only where a
|
||||
#' concept's subtype scope defines what "all" sums over --
|
||||
#' `cog_spending()`/`cog_revenue()`, via `.verb_spendrev()`, pass `TRUE`.
|
||||
#' `cog_balances()` leaves it at the `FALSE` default: holdings are a stock
|
||||
#' with no concept vocabulary to sum across (see R/balances.R), and before
|
||||
#' this guard existed `cog_balances(category = "All Categories")` silently
|
||||
#' matched zero crosswalk rows and returned an empty result with no error
|
||||
#' (finding 3, all-categories review). This validator is shared specifically
|
||||
#' so the three verbs cannot drift apart on this again.
|
||||
#' @noRd
|
||||
.validate_verb_inputs <- function(govid, years, category,
|
||||
per_capita, adjust_to_year, recipe = NULL,
|
||||
allow_all_categories = FALSE) {
|
||||
# NULL is allowed only because the caller has already established that the
|
||||
# cohort is named some other way (`state`/`type`); .make_cohort() is what
|
||||
# refuses a call that names no cohort at all. A supplied-but-empty `govid`
|
||||
# still fails here, exactly as before.
|
||||
if (!is.null(govid) && (!is.character(govid) || length(govid) == 0L)) {
|
||||
per_capita, adjust_to_year, recipe = NULL) {
|
||||
if (!is.character(govid) || length(govid) == 0L) {
|
||||
cli::cli_abort("`govid` must be a non-empty character vector.")
|
||||
}
|
||||
if (!(is.integer(years) || is.numeric(years)) || length(years) == 0L) {
|
||||
@@ -677,14 +484,6 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
if (!is.null(category) && !is.character(category)) {
|
||||
cli::cli_abort("`category` must be character or NULL.")
|
||||
}
|
||||
if (!allow_all_categories && !is.null(category) &&
|
||||
.ALL_CATEGORIES %in% category) {
|
||||
cli::cli_abort(c(
|
||||
"{.val {(.ALL_CATEGORIES)}} is not supported here.",
|
||||
i = "It sums a spending or revenue concept's subtype scope; this verb has no concept vocabulary to sum across.",
|
||||
i = "Use {.fn cog_spending} or {.fn cog_revenue} for an all-categories total."
|
||||
), class = "uscogdata_all_categories_unsupported")
|
||||
}
|
||||
if (!is.logical(per_capita) || length(per_capita) != 1L) {
|
||||
cli::cli_abort("`per_capita` must be a length-1 logical.")
|
||||
}
|
||||
@@ -713,18 +512,6 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
if (identical(basis, "harmonized")) paste0(view_base, "_harmonized") else view_base
|
||||
}
|
||||
|
||||
#' The `*_long`/`*_long_harmonized` view behind an annotated view base --
|
||||
#' `"spending_annotated"` -> `"spending_long_harmonized"`. `.build_suggestions()`
|
||||
#' anti-joins the LONG view rather than the annotated one: they have identical
|
||||
#' row membership (the annotated views are the long views plus LEFT JOINs, see
|
||||
#' inst/sql/42-spending_annotated_harmonized.sql), but the long view is the
|
||||
#' one that actually owns the `NOT is_aggregate` + crosswalk-membership rule
|
||||
#' the suppression test is asking about.
|
||||
#' @noRd
|
||||
.select_long_view <- function(view_base, basis) {
|
||||
.select_view(sub("_annotated$", "_long", view_base), basis)
|
||||
}
|
||||
|
||||
#' @noRd
|
||||
.select_ig_view <- function(basis) {
|
||||
if (identical(basis, "harmonized")) "ig_annotated_harmonized" else "ig_annotated"
|
||||
@@ -769,17 +556,11 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
}
|
||||
|
||||
#' @noRd
|
||||
.build_verb_sql <- function(view, subtype_col, cohort, years, category,
|
||||
ig_view = NULL, subtype_scope = NULL,
|
||||
all_categories = FALSE, limit = NULL, offset = NULL) {
|
||||
cohort_pred <- .cohort_sql(cohort)
|
||||
.build_verb_sql <- function(view, subtype_col, govid, years, category,
|
||||
ig_view = NULL, subtype_scope = NULL) {
|
||||
govid_lit <- .sql_lit_chr(govid)
|
||||
years_lit <- paste(as.integer(years), collapse = ",")
|
||||
# In all-categories mode there is no category filter: the sum is defined by
|
||||
# the concept's SUBTYPE allowlist (subtype_pred below), which is the real
|
||||
# concept boundary. Filtering by category as well would be a no-op at best
|
||||
# and, if the crosswalk ever gained an uncategorized code, a silent
|
||||
# under-count of the very total this mode exists to guarantee.
|
||||
category_pred <- if (all_categories || is.null(category)) {
|
||||
category_pred <- if (is.null(category)) {
|
||||
""
|
||||
} else {
|
||||
sprintf("AND category IN (%s)", .sql_lit_chr(category))
|
||||
@@ -818,59 +599,29 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
# though its dollars came entirely from an aggregate row, silently
|
||||
# suppressing the "Aggregate fallback applied" note on exactly the rows
|
||||
# this feature exists to surface.
|
||||
|
||||
# Collapse the category dimension. subtype is deliberately KEPT: it is what
|
||||
# lets a caller filter the result to `spend_subtype == "operations"` and
|
||||
# get an operating-expenditure total, the measure a fiscal comparison
|
||||
# actually wants. (There is no `subtype` argument -- this is a post-hoc
|
||||
# filter on the returned column, not a query parameter.)
|
||||
category_select <- if (all_categories) {
|
||||
sprintf("%s AS category", .sql_lit_chr(.ALL_CATEGORIES))
|
||||
} else {
|
||||
"category"
|
||||
}
|
||||
category_group <- if (all_categories) "" else ", category"
|
||||
|
||||
base_sql <- sprintf(
|
||||
sprintf(
|
||||
"SELECT
|
||||
year,
|
||||
canonical_govid,
|
||||
COALESCE(xwalk_gov_name, gov_name) AS gov_name,
|
||||
%1$s,
|
||||
%7$s,
|
||||
category,
|
||||
SUM(amt) * 1000.0 AS amt_nominal,
|
||||
string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS codes_included,
|
||||
bool_or(is_aggregate) AS aggregate_fallback
|
||||
FROM %2$s
|
||||
WHERE %3$s
|
||||
WHERE canonical_govid IN (%3$s)
|
||||
AND year IN (%4$s)
|
||||
%5$s
|
||||
%6$s
|
||||
GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, %1$s%8$s
|
||||
ORDER BY year, canonical_govid, %1$s%8$s",
|
||||
subtype_col, source_expr, cohort_pred, years_lit, category_pred, subtype_pred,
|
||||
category_select, category_group
|
||||
GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, %1$s, category
|
||||
ORDER BY year, canonical_govid, %1$s, category",
|
||||
subtype_col, source_expr, govid_lit, years_lit, category_pred, subtype_pred
|
||||
)
|
||||
|
||||
# limit/offset push the page into the query itself instead of pulling every
|
||||
# matching row across the network only to slice and discard most of it
|
||||
# afterward (the pattern behind the 2026-08-06 production incident: a
|
||||
# 193,105-row/194-page sweep re-ran the full query and re-listified every
|
||||
# row on EVERY page). See .paginate_sql() in R/pagination.R for why the
|
||||
# wrapping is an outer SELECT rather than a bare LIMIT on base_sql.
|
||||
.paginate_sql(base_sql, limit, offset)
|
||||
}
|
||||
|
||||
#' Join population onto a result and derive the per-capita columns.
|
||||
#'
|
||||
#' The population lookup is keyed on the govids PRESENT IN `result`, not on the
|
||||
#' cohort that produced it. Those are the only ones the LEFT JOIN below can
|
||||
#' match, so the joined output is identical either way -- but it means this
|
||||
#' works unchanged for a cohort named by predicate (where no id list exists in
|
||||
#' R at all), and on a paginated call it looks up one page's governments
|
||||
#' instead of the whole fleet's.
|
||||
#' @noRd
|
||||
.attach_per_capita <- function(result, con) {
|
||||
.attach_per_capita <- function(result, con, govid) {
|
||||
if (nrow(result) == 0L) {
|
||||
result$amt_per_capita_nominal <- numeric(0)
|
||||
result$pop_source <- character(0)
|
||||
@@ -883,7 +634,7 @@ cog_spending <- function(govid = NULL, years, category = NULL,
|
||||
FROM gov_population_yearly
|
||||
WHERE canonical_govid IN (%s)
|
||||
AND year IN (%s)",
|
||||
.sql_lit_chr(unique(result$canonical_govid)), years_lit
|
||||
.sql_lit_chr(govid), years_lit
|
||||
)
|
||||
pops <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
result <- dplyr::left_join(result, pops,
|
||||
|
||||
+55
-276
@@ -1,16 +1,8 @@
|
||||
# R/suggestions.R
|
||||
# Recipe-component-driven signposting. When a basis = "harmonized" query for
|
||||
# a category comes back incomplete in some requested year -- and a
|
||||
# harmonization recipe would actually fill it for this government -- surface
|
||||
# that recipe as a suggestion. "Incomplete" has two forms, and a recipe
|
||||
# qualifies on either:
|
||||
# 1. empty_year -- the result has no rows at all in that year.
|
||||
# 2. suppressed_component -- the result HAS rows, but a component code
|
||||
# carries dollars the verb's own long view structurally excludes
|
||||
# (aggregate-published, or absent from summary_categories). This is
|
||||
# uscogdata#9: Public Welfare kept returning E74/E79 rows while dropping
|
||||
# aggregate-only E67/E68, so form 1 never fired and the caller got a
|
||||
# number a third too low with no signpost at all.
|
||||
# Recipe-component-driven signposting: when a basis = "harmonized" query for
|
||||
# a category comes back with a coverage gap in some requested years (the
|
||||
# result has no rows at all in that year) that a harmonization recipe would
|
||||
# actually fill for this government, surface that recipe as a suggestion.
|
||||
#
|
||||
# This is deliberately keyed off the recipe catalog's component codes, not
|
||||
# off harmonization_map rows: no live map row carries a non-blank
|
||||
@@ -42,8 +34,8 @@
|
||||
#' verb call: recipes whose generic join would fill a real gap in `result`.
|
||||
#'
|
||||
#' @param con Active DuckDB connection.
|
||||
#' @param cohort The verb's cohort object (see `.make_cohort()`), naming the
|
||||
#' governments by id, by state/type predicate, or both.
|
||||
#' @param govid Character vector of canonical_govid values (the verb's raw
|
||||
#' `govid`).
|
||||
#' @param years Integer vector of requested years.
|
||||
#' @param category `category` argument as passed to the verb (character
|
||||
#' vector or `NULL`; suggestions are only computed when non-NULL).
|
||||
@@ -56,54 +48,11 @@
|
||||
#' "D")` for `cog_revenue()` -- see `.verb_spendrev()`). Passed through to
|
||||
#' `.attach_ig_counterparts()` to keep the intergovernmental-counterpart
|
||||
#' lookup scoped to the calling verb's own flow family.
|
||||
#' @param long_view Name of the verb's own long view (from
|
||||
#' `.select_long_view()`), passed through to `.suppressed_components()` to
|
||||
#' measure the second qualifying path (uscogdata#9).
|
||||
#' @param all_categories `TRUE` when the caller's `category` is the reserved
|
||||
#' pseudo-category (`.ALL_CATEGORIES`). Defaults to `FALSE` so no other
|
||||
#' caller's behaviour changes. When `TRUE`, the candidate-recipe sub-select
|
||||
#' is scoped by `subtype_col`/`subtype_scope` instead of by `category` --
|
||||
#' symmetric with `.build_verb_sql()`'s own all-categories branch (see
|
||||
#' R/spending.R): the concept's subtype allowlist is the real scope
|
||||
#' boundary, not any literal category value, and
|
||||
#' `.ALL_CATEGORIES` ("All Categories") is never itself a row in
|
||||
#' `summary_categories.category`, so leaving the category-keyed sub-select
|
||||
#' in place here always returned zero candidates and silently disabled
|
||||
#' signposting in all-categories mode (final whole-branch review, finding
|
||||
#' 6).
|
||||
#' @param subtype_col Name of the `summary_categories` subtype column to
|
||||
#' scope by when `all_categories = TRUE` (`"spend_subtype"` or
|
||||
#' `"revenue_subtype"` -- the same value `.build_verb_sql()` already
|
||||
#' receives as its own `subtype_col`). Ignored when `all_categories =
|
||||
#' FALSE`. `NULL` by default.
|
||||
#' @param subtype_scope Character vector of subtype values to scope by when
|
||||
#' `all_categories = TRUE` (the same value `.build_verb_sql()` already
|
||||
#' receives as its own `subtype_scope` -- the concept's subtype allowlist,
|
||||
#' e.g. `.expenditure_concept_subtypes(expenditure_concept)`). Ignored when
|
||||
#' `all_categories = FALSE`. `NULL` by default.
|
||||
#' @return List of `list(recipe_id, label, available_years, hint,
|
||||
#' ig_recipe_id, trigger, suppressed_amount, suppressed_years,
|
||||
#' suppressed_codes)`, possibly empty.
|
||||
#'
|
||||
#' Decomposed (Issue #33) into three extracted helpers to stay within the
|
||||
#' project's "functions under 50 lines" convention:
|
||||
#' \itemize{
|
||||
#' \item `.query_candidate_recipes()` -- candidate recipe lookup by
|
||||
#' category/subtype scope + `category_type` filter (#34) + M/L exclusion.
|
||||
#' \item `.query_recipe_meta()` -- metadata (label, year spans).
|
||||
#' \item `.query_covered_years()` -- Path 1 gap-year coverage via the
|
||||
#' recipe's own generic join.
|
||||
#' }
|
||||
#' The for-loop that merges covered-years + suppressed-components into
|
||||
#' suggestion objects stays inline here because it interleaves
|
||||
#' empty_hit/supp_hit precedence with field assembly. Likewise kept inline:
|
||||
#' the M/L-exclusion design-comment block and the final
|
||||
#' `.attach_ig_counterparts()` call.
|
||||
#' ig_recipe_id)`, possibly empty.
|
||||
#' @noRd
|
||||
.build_suggestions <- function(con, cohort, years, category, result, basis,
|
||||
flow_prefixes, long_view,
|
||||
all_categories = FALSE,
|
||||
subtype_col = NULL, subtype_scope = NULL) {
|
||||
.build_suggestions <- function(con, govid, years, category, result, basis,
|
||||
flow_prefixes) {
|
||||
if (!identical(basis, "harmonized") || is.null(category)) return(list())
|
||||
|
||||
# Exclude any recipe that is ITSELF an intergovernmental (M/L) recipe --
|
||||
@@ -118,24 +67,17 @@
|
||||
# flow-prefix gate below/in `.attach_ig_counterparts()`: an M/L recipe
|
||||
# should never be suggested as a coverage-gap filler for EITHER verb, not
|
||||
# just kept from being named as the *counterpart* of another suggestion.
|
||||
#
|
||||
# The inner sub-select is the concept boundary (finding 6, final
|
||||
# whole-branch review): in all-categories mode it is scoped by
|
||||
# `subtype_col`/`subtype_scope` -- the same allowlist `.build_verb_sql()`
|
||||
# applies as a WHERE predicate to make the summed result a *concept*, not
|
||||
# by `category` (`.ALL_CATEGORIES` is never a row in
|
||||
# `summary_categories.category`, so a category-keyed sub-select always
|
||||
# came back empty here). The M/L exclusion below is unchanged either way.
|
||||
#
|
||||
# Issue #34: scope the candidate query by `category_type` ('expenditure'
|
||||
# vs 'revenue') to prevent cross-flow-family leakage -- e.g.
|
||||
# `cog_revenue(category = "Corrections")` must not surface
|
||||
# expenditure-only recipes (E04/E05) merely because they share the same
|
||||
# category name in summary_categories. The type is derived from
|
||||
# flow_prefixes: E/F/G -> 'expenditure', anything else -> 'revenue'.
|
||||
candidates <- .query_candidate_recipes(con, category, flow_prefixes,
|
||||
all_categories, subtype_col,
|
||||
subtype_scope)
|
||||
candidates <- DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||
WHERE component_code IN (
|
||||
SELECT DISTINCT item_code FROM summary_categories WHERE category IN (%s)
|
||||
)
|
||||
AND recipe_id NOT IN (
|
||||
SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||
WHERE LEFT(component_code, 1) IN ('M', 'L')
|
||||
)",
|
||||
.sql_lit_chr(category)
|
||||
))$recipe_id
|
||||
if (length(candidates) == 0L) return(list())
|
||||
|
||||
result_years <- if (is.null(result) || nrow(result) == 0L) {
|
||||
@@ -144,170 +86,22 @@
|
||||
unique(as.integer(result$year))
|
||||
}
|
||||
gap_years <- setdiff(as.integer(years), result_years)
|
||||
if (length(gap_years) == 0L) return(list())
|
||||
|
||||
# Path 2 (uscogdata#9): component dollars this government holds that the
|
||||
# verb's own view structurally excludes. Measured across ALL requested
|
||||
# years, not just gap years -- the whole point is that a year with rows can
|
||||
# still be missing dollars. Scoped to the calling verb's own flow_prefixes
|
||||
# (I1) -- see `.suppressed_components()`'s own roxygen for why.
|
||||
#
|
||||
# This runs unconditionally whenever there are candidates -- an earlier
|
||||
# revision of this fix wave tried a free, in-memory pre-check
|
||||
# (`.needs_suppression_query()`) to skip the round trip on an already-
|
||||
# covered path, but a scoped re-review measured it against the fixture and
|
||||
# found it didn't pay for itself (it skipped ~3% of healthy calls, ~0% of
|
||||
# the multi-govid batch shape it was meant to help, at a net cost increase
|
||||
# once its own always-run metadata query was counted) while adding an
|
||||
# untested exactness invariant -- that `result$codes_included` and this
|
||||
# anti-join share the harmonized `item_code` space -- whose silent
|
||||
# violation would kill signposting, the exact failure class uscogdata#9
|
||||
# exists to prevent. Owner's call: keep this simple; a batch-aware
|
||||
# optimization, if one is worth building, is a separate issue.
|
||||
supp <- .suppressed_components(con, candidates, cohort, years, long_view, flow_prefixes)
|
||||
meta <- tibble::as_tibble(DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT recipe_id, any_value(label) AS label,
|
||||
MIN(year_min) AS year_min, MAX(year_max) AS year_max
|
||||
FROM harmonization_recipes
|
||||
WHERE recipe_id IN (%s)
|
||||
GROUP BY recipe_id",
|
||||
.sql_lit_chr(candidates)
|
||||
)))
|
||||
|
||||
if (length(gap_years) == 0L && nrow(supp) == 0L) return(list())
|
||||
|
||||
meta <- .query_recipe_meta(con, candidates)
|
||||
|
||||
# Path 1 (unchanged): (recipe, year) pairs the recipe's own generic join
|
||||
# covers for this government, restricted to the gap years.
|
||||
covered <- .query_covered_years(con, candidates, cohort, gap_years)
|
||||
|
||||
suggestions <- list()
|
||||
for (rid in candidates) {
|
||||
empty_hit <- rid %in% covered$recipe_id
|
||||
s_rows <- supp[supp$recipe_id == rid, , drop = FALSE]
|
||||
supp_hit <- nrow(s_rows) > 0L
|
||||
if (!empty_hit && !supp_hit) next
|
||||
m <- meta[meta$recipe_id == rid, ]
|
||||
suggestions[[length(suggestions) + 1L]] <- list(
|
||||
recipe_id = rid,
|
||||
label = m$label[[1]],
|
||||
available_years = c(as.integer(m$year_min), as.integer(m$year_max)),
|
||||
hint = sprintf("re-run with recipe = '%s'", rid),
|
||||
# An empty year is the stronger claim -- the category returned nothing
|
||||
# at all -- so it wins when both paths qualify. The suppressed_* fields
|
||||
# are still populated, so an empty_year fire also reports its dollars.
|
||||
trigger = if (empty_hit) "empty_year" else "suppressed_component",
|
||||
suppressed_amount = if (supp_hit) sum(s_rows$suppressed_amount) else 0,
|
||||
suppressed_years = if (supp_hit) {
|
||||
sort(unique(as.integer(s_rows$year)))
|
||||
} else {
|
||||
integer(0)
|
||||
},
|
||||
suppressed_codes = if (supp_hit) {
|
||||
sort(unique(unlist(strsplit(s_rows$suppressed_codes, ",", fixed = TRUE))))
|
||||
} else {
|
||||
character(0)
|
||||
}
|
||||
)
|
||||
}
|
||||
.attach_ig_counterparts(con, suggestions, flow_prefixes)
|
||||
}
|
||||
|
||||
#' Query candidate harmonization recipe IDs for a coverage-gap suggestion.
|
||||
#'
|
||||
#' Selects recipes whose component codes fall within the requested scope
|
||||
#' (category or subtype allowlist), excluding any recipe that is ITSELF an
|
||||
#' intergovernmental (M/L) recipe -- i.e. every one of its own component
|
||||
#' codes is M/L-prefixed. Without this exclusion, a category whose
|
||||
#' summary_categories rows span both a Direct family (e.g. E04/E05,
|
||||
#' "Corrections") and its M/L counterpart (M04/M05) makes the M/L recipe
|
||||
#' itself a raw top-level candidate for a plain `cog_spending()` call --
|
||||
#' following that hint would silently return intergovernmental dollars
|
||||
#' under `expenditure_concept = "direct"` provenance.
|
||||
#'
|
||||
#' In all-categories mode (`all_categories = TRUE`) the inner sub-select is
|
||||
#' scoped by `subtype_col`/`subtype_scope` -- the same allowlist
|
||||
#' `.build_verb_sql()` applies as a WHERE predicate to make the summed
|
||||
#' result a *concept* (see R/spending.R), not by `category`.
|
||||
#' `.ALL_CATEGORIES` ("All Categories") is never itself a row in
|
||||
#' `summary_categories.category`, so a category-keyed sub-select always
|
||||
#' returns zero candidates and silently disables signposting.
|
||||
#'
|
||||
#' Scope is also by `category_type` ('expenditure' vs 'revenue', Issue #34)
|
||||
#' to prevent cross-flow-family leakage: `cog_revenue(category =
|
||||
#' "Corrections")` must not surface expenditure-only recipes (E04/E05)
|
||||
#' merely because they share the same category name in summary_categories.
|
||||
#' The type is derived from flow_prefixes: E/F/G -> 'expenditure', anything
|
||||
#' else -> 'revenue'.
|
||||
#'
|
||||
#' @param con Active DuckDB connection.
|
||||
#' @param category Category name, or `NULL`.
|
||||
#' @param flow_prefixes The calling verb's own flow-type prefixes (see
|
||||
#' `.build_suggestions()`). Used to derive `category_type` (#34).
|
||||
#' @param all_categories `TRUE` when the caller used `.ALL_CATEGORIES`.
|
||||
#' @param subtype_col Name of the summary_categories subtype column to
|
||||
#' scope by when `all_categories = TRUE`; ignored otherwise.
|
||||
#' @param subtype_scope Character vector of subtype values to scope by
|
||||
#' when `all_categories = TRUE`; ignored otherwise.
|
||||
#' @return Character vector of recipe IDs (possibly empty).
|
||||
#' @noRd
|
||||
.query_candidate_recipes <- function(con, category, flow_prefixes,
|
||||
all_categories = FALSE,
|
||||
subtype_col = NULL,
|
||||
subtype_scope = NULL) {
|
||||
# Issue #34: derive category_type from flow_prefixes to prevent
|
||||
# cross-flow-family leakage -- e.g. cog_revenue(category = "Corrections")
|
||||
# must not surface expenditure-only recipes merely because they share the
|
||||
# same category name in summary_categories.
|
||||
category_type <- if (all(flow_prefixes %in% c("E", "F", "G"))) {
|
||||
"expenditure"
|
||||
} else {
|
||||
"revenue"
|
||||
}
|
||||
|
||||
candidate_scope_sql <- if (isTRUE(all_categories)) {
|
||||
sprintf(
|
||||
"SELECT DISTINCT item_code FROM summary_categories
|
||||
WHERE %s IN (%s) AND category_type = '%s'",
|
||||
subtype_col, .sql_lit_chr(subtype_scope), category_type
|
||||
)
|
||||
} else {
|
||||
sprintf(
|
||||
"SELECT DISTINCT item_code FROM summary_categories
|
||||
WHERE category IN (%s) AND category_type = '%s'",
|
||||
.sql_lit_chr(category), category_type
|
||||
)
|
||||
}
|
||||
|
||||
DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||
WHERE component_code IN (
|
||||
%s
|
||||
)
|
||||
AND recipe_id NOT IN (
|
||||
SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||
WHERE LEFT(component_code, 1) IN ('M', 'L')
|
||||
)",
|
||||
candidate_scope_sql
|
||||
))$recipe_id
|
||||
}
|
||||
|
||||
#' Query gap-year coverage: which (recipe, year) pairs the recipe's own
|
||||
#' generic join covers for this government, restricted to `gap_years`.
|
||||
#'
|
||||
#' This is Path 1 of a suggestion (unchanged): it finds recipes whose
|
||||
#' component codes' generic join produces at least one row for this
|
||||
#' government in each gap year -- i.e. the category returned nothing in
|
||||
#' that year but a recipe would fill it.
|
||||
#'
|
||||
#' @param con Active DuckDB connection.
|
||||
#' @param candidates Character vector of recipe IDs to check coverage for.
|
||||
#' @param cohort The verb's cohort object (see `.make_cohort()`), rendered
|
||||
#' into the govid predicate on the joined `long` scan via `.cohort_sql()`.
|
||||
#' @param gap_years Integer vector of requested years absent from the
|
||||
#' result.
|
||||
#' @return Data frame with columns `recipe_id` (character) and `year`
|
||||
#' (integer). Returns an empty data frame (`recipe_id = character(0)`,
|
||||
#' `year = integer(0)`) when `gap_years` is empty, so callers can safely
|
||||
#' reference `$recipe_id`.
|
||||
#' @noRd
|
||||
.query_covered_years <- function(con, candidates, cohort, gap_years) {
|
||||
if (length(gap_years) == 0L) {
|
||||
return(data.frame(recipe_id = character(0), year = integer(0)))
|
||||
}
|
||||
DBI::dbGetQuery(con, sprintf(
|
||||
# Which (recipe_id, year) pairs the recipe's own generic join actually
|
||||
# covers for this government, restricted to the gap years -- the same
|
||||
# join .run_recipe() uses (component year_min/year_max + gov_type_scope,
|
||||
# no is_aggregate filter), just checking existence instead of summing.
|
||||
covered <- DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT DISTINCT r.recipe_id, l.year
|
||||
FROM long l
|
||||
JOIN harmonization_recipes r
|
||||
@@ -317,30 +111,24 @@
|
||||
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||
WHERE r.recipe_id IN (%s)
|
||||
AND %s
|
||||
AND l.canonical_govid IN (%s)
|
||||
AND l.year IN (%s)",
|
||||
.sql_lit_chr(candidates), .cohort_sql(cohort, "l.canonical_govid"),
|
||||
.sql_lit_chr(candidates), .sql_lit_chr(govid),
|
||||
paste(gap_years, collapse = ",")
|
||||
))
|
||||
}
|
||||
|
||||
#' Query recipe metadata: labels and year spans for a set of candidate
|
||||
#' recipes.
|
||||
#'
|
||||
#' @param con Active DuckDB connection.
|
||||
#' @param candidates Character vector of recipe IDs to look up.
|
||||
#' @return Tibble with columns `recipe_id`, `label`, `year_min` (int), and
|
||||
#' `year_max` (int).
|
||||
#' @noRd
|
||||
.query_recipe_meta <- function(con, candidates) {
|
||||
tibble::as_tibble(DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT recipe_id, any_value(label) AS label,
|
||||
MIN(year_min) AS year_min, MAX(year_max) AS year_max
|
||||
FROM harmonization_recipes
|
||||
WHERE recipe_id IN (%s)
|
||||
GROUP BY recipe_id",
|
||||
.sql_lit_chr(candidates)
|
||||
)))
|
||||
suggestions <- list()
|
||||
for (rid in candidates) {
|
||||
if (!rid %in% covered$recipe_id) next
|
||||
m <- meta[meta$recipe_id == rid, ]
|
||||
suggestions[[length(suggestions) + 1L]] <- list(
|
||||
recipe_id = rid,
|
||||
label = m$label[[1]],
|
||||
available_years = c(as.integer(m$year_min), as.integer(m$year_max)),
|
||||
hint = sprintf("re-run with recipe = '%s'", rid)
|
||||
)
|
||||
}
|
||||
.attach_ig_counterparts(con, suggestions, flow_prefixes)
|
||||
}
|
||||
|
||||
#' Attach `ig_recipe_id` to each suggestion: the intergovernmental-expenditure
|
||||
@@ -388,7 +176,7 @@
|
||||
#' `R/basis.R`). This blocks a recipe surfaced through a mis-scoped
|
||||
#' category from ever reaching the M/L search, e.g. `cog_spending()`'s
|
||||
#' flow_prefixes are `c("E","F","G")`, which `ig_federal_b47_wide`'s own
|
||||
#' "B" is not part of.
|
||||
#' `"B"` is not part of.
|
||||
#' 2. `own_prefix %in% c("E","F","G")`: M/L only ever pairs with the
|
||||
#' DIRECT-expenditure family, never with revenue (`cog_revenue()`'s
|
||||
#' flow_prefixes already fold B/C/D in as ordinary revenue -- there is
|
||||
@@ -396,6 +184,10 @@
|
||||
#' adds one for spending) and never with ANOTHER M/L recipe (without
|
||||
#' this check, `ige_local_m47_wide` would wrongly match sibling
|
||||
#' `ige_state_l47_wide` on their shared {"47","94"} suffix set).
|
||||
#' Condition 1 alone does not catch this: under `cog_revenue()`,
|
||||
#' `ig_federal_b47_wide`'s own `"B"` IS inside revenue's own
|
||||
#' `flow_prefixes`, so only this second, family-specific check blocks
|
||||
#' the search.
|
||||
#' @noRd
|
||||
.attach_ig_counterparts <- function(con, suggestions, flow_prefixes) {
|
||||
if (length(suggestions) == 0L) return(suggestions)
|
||||
@@ -440,25 +232,12 @@
|
||||
#' expressions. When a suggestion has an `ig_recipe_id`, one indented
|
||||
#' continuation line is appended naming the intergovernmental counterpart
|
||||
#' recipe (embedded `\n` renders as a hanging-indent continuation of the
|
||||
#' same bullet under cli, not a new bullet). Same treatment for
|
||||
#' `suppressed_amount` (uscogdata#9): only present when dollars were
|
||||
#' actually measured as excluded (an `empty_year` fire can carry them too --
|
||||
#' see `.build_suggestions()` -- so this keys off the amount, not `trigger`).
|
||||
#' same bullet under cli, not a new bullet).
|
||||
#' @noRd
|
||||
.inform_suggestions <- function(suggestions) {
|
||||
bullets <- vapply(suggestions, function(s) {
|
||||
bullet <- sprintf("%s (%d-%d): %s", s$recipe_id,
|
||||
s$available_years[1], s$available_years[2], s$hint)
|
||||
# Only present when dollars were actually measured as excluded. An
|
||||
# empty_year fire can carry them too -- the year had no rows AND the
|
||||
# component was suppressed -- which is strictly more informative.
|
||||
if (isTRUE(s$suppressed_amount > 0)) {
|
||||
bullet <- paste0(bullet, sprintf(
|
||||
"\n $%s excluded from %s (%s), published as an aggregate or outside the crosswalk",
|
||||
formatC(s$suppressed_amount, format = "f", digits = 0, big.mark = ","),
|
||||
paste0("FY", s$suppressed_years, collapse = ", "),
|
||||
paste(s$suppressed_codes, collapse = ", ")))
|
||||
}
|
||||
if (!is.null(s$ig_recipe_id)) {
|
||||
bullet <- paste0(bullet, sprintf(
|
||||
"\n intergovernmental counterpart: recipe = '%s'", s$ig_recipe_id))
|
||||
@@ -466,7 +245,7 @@
|
||||
bullet
|
||||
}, character(1))
|
||||
cli::cli_inform(c(
|
||||
i = "Incomplete coverage for the requested years; a harmonization recipe may fill it:",
|
||||
i = "Coverage gap detected for the requested years; a harmonization recipe may fill it:",
|
||||
stats::setNames(bullets, rep("*", length(bullets)))
|
||||
))
|
||||
}
|
||||
|
||||
-117
@@ -1,117 +0,0 @@
|
||||
# R/suppression.R
|
||||
# Split out of R/suggestions.R (2026-08-05) to keep files under the project's
|
||||
# 400-line limit. Owns the second qualifying path for coverage signposting
|
||||
# (uscogdata#9): measuring, per government, the component dollars the
|
||||
# calling verb's own long view structurally excludes (aggregate-published,
|
||||
# or absent from summary_categories). See R/suggestions.R for the
|
||||
# orchestrator (`.build_suggestions()`) that calls this and the full
|
||||
# uscogdata#9 background.
|
||||
|
||||
#' Measure, per (recipe, year), the component dollars this government holds
|
||||
#' that the calling verb's own long view structurally excludes.
|
||||
#'
|
||||
#' This is the second qualifying path for a suggestion (uscogdata#9). The
|
||||
#' first -- row absence -- only fires when a category returns NOTHING in a
|
||||
#' requested year, which is how Corrections behaves in the wide era. Public
|
||||
#' Welfare is the failure mode it misses: E74/E75/E77/E79 still return rows,
|
||||
#' so there is no absence to detect, while E67/E68 (aggregate-flagged 1967-
|
||||
#' 2011, and absent from `summary_categories` entirely) are dropped. The
|
||||
#' caller gets a plausible number a third too low, silently.
|
||||
#'
|
||||
#' "Structurally excluded" is decided by anti-joining the verb's REAL long
|
||||
#' view rather than restating its WHERE clause, so this stays correct if
|
||||
#' `spending_long_harmonized` / `revenue_long_harmonized` ever change. That
|
||||
#' anti-join is keyed on `item_code`, which is sound only because
|
||||
#' harmonization never renames a recipe component -- asserted by the "no
|
||||
#' recipe component is ever renamed by harmonization" test in
|
||||
#' tests/testthat/test-recipes.R.
|
||||
#'
|
||||
#' Note what this deliberately does NOT count as suppressed: a component
|
||||
#' excluded from the RESULT for scoping reasons -- because it belongs to a
|
||||
#' different `category`, or because `expenditure_concept` narrowed the
|
||||
#' subtypes -- is still present in the view, so it never fires. Suggesting a
|
||||
#' recipe is a coverage fix, not a category redefinition.
|
||||
#'
|
||||
#' `flow_prefixes` (uscogdata#9 review, finding I1) restricts the measured
|
||||
#' components to the CALLING VERB's own flow family (`c("E","F","G")` for
|
||||
#' spending, `c("T","A","U","B","C","D")` for revenue). Without this, a
|
||||
#' candidate recipe belonging to the OTHER flow family is always absent from
|
||||
#' this verb's view (by construction -- `cog_revenue()`'s view never carries
|
||||
#' an E-coded row) and so was always reported as "suppressed", fabricating a
|
||||
#' dollar claim across flow families (`cog_revenue(category = "Corrections")`
|
||||
#' claimed $3.63B excluded that `cog_spending()` reports and fully accounts
|
||||
#' for). Filtering on `LEFT(r.component_code, 1)` also drops M/L-prefixed
|
||||
#' components from measurement under `cog_spending()` (`flow_prefixes` never
|
||||
#' includes "M"/"L") -- harmless today, because a recipe's own M/L components
|
||||
#' (e.g. `corrections_ig_local_combined`'s M04/M05) are present in the view
|
||||
#' in every year they exist and so never fired as suppressed anyway, but
|
||||
#' worth recording since this filter is now the thing relied on to prevent
|
||||
#' it.
|
||||
#'
|
||||
#' @param con Active DuckDB connection.
|
||||
#' @param candidates Character vector of recipe ids to measure.
|
||||
#' @param cohort The verb's cohort object (see `.make_cohort()`), rendered
|
||||
#' into the govid predicate on both the outer scan and the restated
|
||||
#' NOT EXISTS filter.
|
||||
#' @param years Integer vector of requested years.
|
||||
#' @param long_view Name of the verb's long view, from `.select_long_view()`.
|
||||
#' @param flow_prefixes The calling verb's own flow-type prefixes (see
|
||||
#' `.build_suggestions()`). Only recipe components whose first character is
|
||||
#' in this set are measured.
|
||||
#' @return Tibble of `recipe_id`, `year`, `suppressed_amount` (full US
|
||||
#' dollars), `suppressed_codes` (comma-joined, sorted). Zero rows when
|
||||
#' nothing is suppressed.
|
||||
#' @noRd
|
||||
.suppressed_components <- function(con, candidates, cohort, years, long_view,
|
||||
flow_prefixes) {
|
||||
empty <- tibble::tibble(
|
||||
recipe_id = character(0), year = numeric(0),
|
||||
suppressed_amount = numeric(0), suppressed_codes = character(0)
|
||||
)
|
||||
if (length(candidates) == 0L) return(empty)
|
||||
|
||||
# long_view is interpolated as a SQL IDENTIFIER, not a literal, so it can
|
||||
# never be quoted safely. It is always internally derived from a fixed
|
||||
# view_base, so an off-allowlist value is a programming error, not input.
|
||||
if (!long_view %in% c("spending_long", "spending_long_harmonized",
|
||||
"revenue_long", "revenue_long_harmonized")) {
|
||||
cli::cli_abort(
|
||||
"Internal error: unexpected `long_view` {.val {long_view}}.",
|
||||
class = "uscogdata_internal_error"
|
||||
)
|
||||
}
|
||||
|
||||
sql <- sprintf(
|
||||
"SELECT r.recipe_id,
|
||||
l.year,
|
||||
SUM(l.amt) * 1000.0 AS suppressed_amount,
|
||||
string_agg(DISTINCT l.item_code, ',' ORDER BY l.item_code)
|
||||
AS suppressed_codes
|
||||
FROM long l
|
||||
JOIN harmonization_recipes r
|
||||
ON l.item_code = r.component_code
|
||||
AND l.year BETWEEN r.year_min AND r.year_max
|
||||
AND (r.gov_type_scope = 'all'
|
||||
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||
WHERE r.recipe_id IN (%1$s)
|
||||
AND %2$s
|
||||
AND l.year IN (%3$s)
|
||||
AND l.amt <> 0
|
||||
AND LEFT(r.component_code, 1) IN (%5$s)
|
||||
AND NOT EXISTS (
|
||||
SELECT 1 FROM %4$s v
|
||||
WHERE v.canonical_govid = l.canonical_govid
|
||||
AND v.year = l.year
|
||||
AND v.item_code = l.item_code
|
||||
AND v.year IN (%3$s) -- restated: enables partition pruning (I3a)
|
||||
AND %6$s -- restated: pushes the cohort filter (I3a)
|
||||
)
|
||||
GROUP BY 1, 2
|
||||
ORDER BY 1, 2",
|
||||
.sql_lit_chr(candidates), .cohort_sql(cohort, "l.canonical_govid"),
|
||||
paste(as.integer(years), collapse = ","), long_view,
|
||||
.sql_lit_chr(flow_prefixes), .cohort_sql(cohort, "v.canonical_govid")
|
||||
)
|
||||
tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
}
|
||||
@@ -78,71 +78,6 @@
|
||||
file %in% basename(paths)
|
||||
}
|
||||
|
||||
#' Build the SQL path expression for the partitioned `long` table.
|
||||
#'
|
||||
#' DuckDB cannot expand a glob over generic HTTP: there is no directory
|
||||
#' listing to expand against, and `allow_asterisks_in_http_paths` only
|
||||
#' forwards the literal `**/*` as a filename, which 404s. Measured against
|
||||
#' the published corpus on 2026-08-08, an explicit file list returns the
|
||||
#' same 46,148,034 rows the (working) `hf://` glob does, and
|
||||
#' `hive_partitioning = true` still recovers `year` from the paths.
|
||||
#'
|
||||
#' The manifest already enumerates every partition, so we build the list
|
||||
#' from it. This is host-agnostic -- Nextcloud, HuggingFace and a local
|
||||
#' fixture take the same path -- where an `hf://` URL would tie the reader
|
||||
#' to one vendor's protocol and still need special-casing, since manifest
|
||||
#' fetching goes through httr2, which cannot speak `hf://`.
|
||||
#'
|
||||
#' Falls back to the glob when the manifest carries no partition list: a
|
||||
#' hand-built manifest in a test (see test-views.R) or a corpus predating
|
||||
#' the field. Both are local, where globbing works.
|
||||
#' @noRd
|
||||
.long_files_sql <- function(url, manifest) {
|
||||
parts <- manifest$files$long_partitions %||% list()
|
||||
if (length(parts) == 0L) {
|
||||
return(.sql_lit_chr(paste0(url, "data/long/**/*.parquet")))
|
||||
}
|
||||
paths <- vapply(parts, function(p) as.character(p$path), character(1))
|
||||
paste0("[", .sql_lit_chr(paste0(url, paths)), "]")
|
||||
}
|
||||
|
||||
#' Substitute the corpus-location tokens in a view's SQL text.
|
||||
#'
|
||||
#' One place knows the token vocabulary. `.register_views()` and the tests
|
||||
#' that execute a view file directly both route through here. This exists
|
||||
#' because four test sites had hand-rolled the `{url}` substitution -- one
|
||||
#' of them commented as doing it "exactly as .register_views() does" -- and
|
||||
#' every one of them broke the moment a second token was introduced.
|
||||
#'
|
||||
#' `{long_files}` must be substituted BEFORE `{url}`: it expands to a string
|
||||
#' that itself contains the url, so the reverse order leaves the token in
|
||||
#' place and DuckDB's parser fails on the brace.
|
||||
#'
|
||||
#' `manifest` defaults to empty, which routes `.long_files_sql()` to its glob
|
||||
#' fallback -- correct for the local temp corpora the direct-execution tests
|
||||
#' build.
|
||||
#' @noRd
|
||||
#' `fixed = TRUE` is load-bearing, not a style choice.
|
||||
#'
|
||||
#' In regex mode, `gsub()` interprets backslashes in the REPLACEMENT string as
|
||||
#' escape sequences and silently drops them. A Windows corpus path is full of
|
||||
#' them, so `C:\Users\RUNNER\AppData\...` was substituted in as
|
||||
#' `C:UsersRUNNERAppData...` and every DuckDB read failed with "No files found
|
||||
#' that match the pattern". `fixed = TRUE` treats pattern and replacement as
|
||||
#' literal text, which is what a filesystem path needs.
|
||||
#'
|
||||
#' This is why the package could not read a LOCAL corpus on Windows at all --
|
||||
#' including the test fixture, hence the entire suite, and any `cog_mirror()`
|
||||
#' copy. Remote https URLs were unaffected, having no backslashes, which is
|
||||
#' part of why it stayed hidden: the bug predates the `{long_files}` token and
|
||||
#' lived in the original `{url}` substitution, unnoticed because nothing ever
|
||||
#' ran on Windows until the mirror's check matrix existed.
|
||||
#' @noRd
|
||||
.render_view_sql <- function(sql, url, manifest = list()) {
|
||||
sql <- gsub("{long_files}", .long_files_sql(url, manifest), sql, fixed = TRUE)
|
||||
gsub("{url}", url, sql, fixed = TRUE)
|
||||
}
|
||||
|
||||
#' Register DuckDB views from inst/sql/ SQL files
|
||||
#' @noRd
|
||||
.register_views <- function(con, url, manifest) {
|
||||
@@ -156,7 +91,7 @@
|
||||
!.corpus_has_table(manifest, .representation_view_files[[base]])) next
|
||||
if (base %in% .balance_view_files && !.corpus_has_balance_subtype(con)) next
|
||||
sql <- paste(readLines(f, warn = FALSE), collapse = "\n")
|
||||
sql <- .render_view_sql(sql, url, manifest)
|
||||
sql <- gsub("\\{url\\}", url, sql, fixed = FALSE)
|
||||
DBI::dbExecute(con, sql)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,152 +1,24 @@
|
||||
# uscogdata
|
||||
|
||||
<!-- badges: start -->
|
||||
[](https://github.com/civilytics/uscogdata/actions/workflows/R-CMD-check.yaml)
|
||||
[](https://civilytics.r-universe.dev/uscogdata)
|
||||
[](LICENSE.md)
|
||||
<!-- badges: end -->
|
||||
Curated R reader for the Civilytics US Census of Governments finance corpus.
|
||||
|
||||
A curated R reader for the Civilytics US Census of Governments finance corpus —
|
||||
every dollar that US state, county, municipal and township governments reported
|
||||
raising and spending, from **FY1967 to FY2024**, in one queryable place.
|
||||
Provides unit-level financial profiles, geographic rollups, and peer comparisons
|
||||
with auditable provenance and built-in cross-vintage correctness. Reads the
|
||||
published corpus (Hive-partitioned parquet + manifest.json) directly from
|
||||
Nextcloud via DuckDB httpfs — no local bulk downloads required.
|
||||
|
||||
The Census of Governments is the only nationwide source for local government
|
||||
finance, and it is hard to use: item codes change meaning across vintages,
|
||||
government identifiers were renumbered in 2017, and an absent value means
|
||||
"published zero" in one era and "not reported" in the next. This package
|
||||
handles each of those problems, and it tells you when it has — every result
|
||||
carries provenance describing what was converted, what was aggregated, and
|
||||
which known series breaks intersect your query.
|
||||
## Status
|
||||
|
||||
**Scope:** government types 0–3 (state, county, municipality, township).
|
||||
56 fiscal years, 46,148,034 rows, ~201 MB. There is no source data for FY1968
|
||||
or FY1969. Special districts (type 4) and school districts (type 5) are
|
||||
excluded pending validation.
|
||||
Under active development (Phase 2 of the cog_pipeline project). See
|
||||
`../cog_pipeline/docs/reader-specification.md` for the reader contract this
|
||||
package implements.
|
||||
|
||||
## Where the data comes from
|
||||
|
||||
The corpus is published and documented at the **[US Census of Governments
|
||||
Finance API](https://pages.civilytics.org/cog-api/)**. Start there for how the
|
||||
data was built, how the identifier and item-code reconciliation works, and what
|
||||
the corpus does and does not cover.
|
||||
|
||||
- **[API documentation and walkthroughs](https://pages.civilytics.org/cog-api/)**
|
||||
— reference, data dictionary, and worked examples such as the
|
||||
[Southern states guide](https://pages.civilytics.org/cog-api/cog-api-south-guide.html)
|
||||
- **[Live API](https://cog-api.civilytics.org/api/v1/)** — the same corpus over
|
||||
HTTP, for Tableau, Python, or anything that isn't R
|
||||
- **[Bulk corpus on Hugging Face](https://huggingface.co/datasets/civilytics/us-cog-finance)**
|
||||
— CC-BY-4.0; the same parquet files this package reads
|
||||
- **[Census Bureau source data](https://www.census.gov/programs-surveys/gov-finances.html)**
|
||||
— the underlying public files
|
||||
|
||||
## Install
|
||||
## Installation
|
||||
|
||||
```r
|
||||
install.packages("uscogdata",
|
||||
repos = c("https://civilytics.r-universe.dev",
|
||||
"https://cloud.r-project.org"))
|
||||
# pak::pkg_install("gitea.civilytics.org/Civilytics/uscogdata")
|
||||
```
|
||||
|
||||
Or from source:
|
||||
|
||||
```r
|
||||
pak::pkg_install("git::https://gitea.civilytics.org/Civilytics/uscogdata.git")
|
||||
```
|
||||
|
||||
## Quickstart
|
||||
|
||||
No configuration, no credentials, no download. The package reads the published
|
||||
corpus over HTTPS by default.
|
||||
|
||||
```r
|
||||
library(uscogdata)
|
||||
|
||||
# Resolve a place name to a canonical government id
|
||||
madison <- cog_gov_search(name = "Madison", state = "WI", type = 2)
|
||||
madison$canonical_govid
|
||||
#> [1] "552025209777"
|
||||
|
||||
# Police spending, inflation-adjusted and per capita
|
||||
spend <- cog_spending(
|
||||
madison$canonical_govid,
|
||||
years = 2012:2022,
|
||||
category = "Police",
|
||||
per_capita = TRUE,
|
||||
adjust_to_year = 2023
|
||||
)
|
||||
|
||||
# What did that result do to the numbers, and what should you know about them?
|
||||
cog_explain(spend)
|
||||
```
|
||||
|
||||
`years` is required — there is no implicit full-history default.
|
||||
|
||||
## Two ways to read the corpus
|
||||
|
||||
| | Remote (default) | Mirrored |
|
||||
|---|---|---|
|
||||
| Setup | none | `cog_mirror(dest)`, ~201 MB once |
|
||||
| Disk used | **0 MB** — HTTP range requests only | ~201 MB |
|
||||
| Opening a session | ~7.5 s | ~0.1 s |
|
||||
| One government, one year | ~4 s | ~0.05 s |
|
||||
| One government, full history | ~7 s | ~0.1 s |
|
||||
| Later queries, same session | ~1.5 s | ~0.05 s |
|
||||
| Good for | trying it out, teaching, one-off questions | repeated analysis, offline work, reproducibility |
|
||||
|
||||
**A local mirror is roughly 60–80x faster, and it is one function call.** That is
|
||||
by far the largest difference any of these settings makes. If you are going to
|
||||
ask more than a handful of questions, mirror first.
|
||||
|
||||
Measured 2026-08-10 on a 16-core Linux workstation against the published corpus
|
||||
(schema v7, `pipeline_commit 3d28ddd`), fresh R session per arm. A one-off
|
||||
question costs about **12 seconds end to end remotely and 0.15 seconds
|
||||
mirrored**, session setup included.
|
||||
|
||||
Two things the per-query rows hide:
|
||||
|
||||
- **Opening the session is the single largest remote cost** — larger than any
|
||||
one query. It fetches the manifest and registers 23 SQL views over HTTPS, and
|
||||
it lands on your first query, not on `library(uscogdata)`.
|
||||
- **The cost is network round-trips, not scanning.** A repeat query against
|
||||
partitions this session has already touched is ~1.5 s rather than ~4 s, and a
|
||||
full-history query costs ~7 s whether it runs first or last. What you are
|
||||
paying for is reaching each of the 56 yearly files over HTTPS the first time.
|
||||
|
||||
Nothing is written to disk in remote mode: DuckDB fetches the parquet footer,
|
||||
works out which row groups it needs, and reads only those. Nothing is cached
|
||||
between sessions either, so every query goes back to the network — and a session
|
||||
that issues many remote queries in quick succession can be rate-limited by the
|
||||
host (`HTTP Error: ... 429`). Both are further reasons to mirror for real work.
|
||||
|
||||
The default points at a public HuggingFace mirror of the corpus. If you would
|
||||
rather not depend on a third party — for reproducibility, for an air-gapped
|
||||
environment, or on principle — **the escape hatch is one function call**:
|
||||
|
||||
```r
|
||||
cog_mirror("~/cog-corpus")
|
||||
Sys.setenv(USCOGDATA_URL = "~/cog-corpus/")
|
||||
```
|
||||
|
||||
After that, nothing in your analysis touches an external service.
|
||||
|
||||
### Configuration
|
||||
|
||||
- `USCOGDATA_URL` — corpus root: an HTTPS URL or a local path, **trailing slash required**
|
||||
- `USCOGDATA_CACHE_DIR` — where the manifest is cached (default: user cache dir)
|
||||
- `USCOGDATA_MANIFEST_TTL_SECS` — manifest re-fetch interval (default 3600)
|
||||
- `USCOGDATA_DUCKDB_THREADS` — cap DuckDB's thread count (default: every visible core)
|
||||
- `USCOGDATA_DUCKDB_MEMORY_LIMIT` — cap DuckDB's memory, e.g. `"4GB"` (default: DuckDB's own)
|
||||
|
||||
Each also has an `options()` spelling — `uscogdata.url`, `uscogdata.duckdb_threads`,
|
||||
and so on — and the environment variable wins where both are set.
|
||||
|
||||
The two DuckDB caps exist for **servers**, not laptops. Unset, DuckDB claims every
|
||||
core it can see, which is right for one interactive session on your own machine and
|
||||
wrong when several readers share a box: each claims the whole machine and they fight.
|
||||
Capping costs roughly 5% on a single query and is worth it anywhere the process is
|
||||
sharing hardware.
|
||||
|
||||
## Amounts are in full US dollars
|
||||
|
||||
Every amount column this package returns — `amt_nominal`, `amt_real`,
|
||||
@@ -157,127 +29,111 @@ own `amt` column preserves that. The verbs multiply by 1000 on the way out, so
|
||||
you never have to. The conversion is recorded in every result:
|
||||
|
||||
```r
|
||||
attr(spend, "provenance")$transformations$units_conversion
|
||||
#> $applied TRUE
|
||||
#> $source_unit "$1,000s (raw Census)"
|
||||
#> $target_unit "$USD"
|
||||
#> $multiplier 1000
|
||||
r <- cog_spending("552025209777", 2020L)
|
||||
attr(r, "provenance")$transformations$units_conversion
|
||||
#> $applied TRUE $source_unit "$1,000s (raw Census)" $target_unit "$USD" $multiplier 1000
|
||||
```
|
||||
|
||||
**Do not multiply again.** If you have read elsewhere that COG amounts are in
|
||||
`$1,000s` — which is true of the raw Census files and of the corpus's own `amt`
|
||||
column — that rule does not apply to anything a `cog_*()` verb hands you.
|
||||
Applying it twice overstates every figure by 1000x, and the result looks
|
||||
plausible rather than obviously wrong.
|
||||
`$1,000s` — true of the raw corpus, and of `cog_explorer`'s conventions doc —
|
||||
that rule does not apply to anything a `cog_*()` verb hands you. Applying it
|
||||
twice overstates every figure by 1000x, and the result looks plausible rather
|
||||
than obviously wrong.
|
||||
|
||||
## Concepts worth understanding before you publish a number
|
||||
## Configuration
|
||||
|
||||
### Primary vs Direct vs Total spending
|
||||
- `USCOGDATA_URL` — corpus root URL (public Nextcloud share, trailing slash)
|
||||
- `USCOGDATA_CACHE_DIR` — optional override for the manifest cache directory
|
||||
- `USCOGDATA_MANIFEST_TTL_SECS` — optional manifest re-fetch TTL (default 3600)
|
||||
|
||||
## Primary vs Direct vs Total spending
|
||||
|
||||
`cog_spending(..., expenditure_concept = c("primary", "direct", "total"))`
|
||||
controls *whose* spending a result counts. Concepts are defined as sets of the
|
||||
crosswalk's `spend_subtype` values, never item-code first letters — the letter
|
||||
`Y` alone spans revenue, expenditure and balance codes.
|
||||
controls whose spending a result counts. Concepts are defined as sets of the
|
||||
crosswalk's `spend_subtype` values — never item-code first letters, which
|
||||
cannot classify correctly (the letter `Y` alone spans revenue, expenditure,
|
||||
and balance codes):
|
||||
|
||||
- **`"primary"`** (default) — the government's own service provision: current
|
||||
operations, capital outlay, assistance payments.
|
||||
- **`"direct"`** — Census's published Direct Expenditure: `primary` plus
|
||||
interest on debt and insurance trust benefits (e.g. pensions).
|
||||
- **`"total"`** — adds the intergovernmental leg, money handed to other
|
||||
governments to spend. Meaningful for one government's own budget over time,
|
||||
but it double-counts when summed across governments: a state's payment to a
|
||||
county is the same dollar the county reports as its own direct spending.
|
||||
- `"primary"` (the default) is the government's own service provision:
|
||||
current operations, capital outlay, and assistance payments.
|
||||
- `"direct"` is Census's published Direct Expenditure: `primary` plus
|
||||
interest on debt and insurance trust benefit payments (e.g. pensions).
|
||||
- `"total"` additionally adds the intergovernmental leg — money handed to
|
||||
other governments to spend (`M`/`L` codes plus `Q11`/`Q12`/`Q18` state
|
||||
payments to school systems) — which is meaningful for describing one
|
||||
government's own budget over time, but double-counts when summed across
|
||||
governments (a state's payment to a county is the same dollar the county
|
||||
reports as its own direct spending).
|
||||
|
||||
**Rule of thumb: any figure spanning more than one government uses `primary`
|
||||
or `direct`.** `cog_geographic_rollup()` and `cog_peer_compare()` enforce that
|
||||
by refusing `"total"` outright. Worked examples in
|
||||
`vignette("total-spending", package = "uscogdata")`.
|
||||
**Rule of thumb: any figure that spans more than one government uses
|
||||
`primary` or `direct`.** `cog_geographic_rollup()` and `cog_peer_compare()`
|
||||
enforce this by refusing `expenditure_concept = "total"`. See
|
||||
`vignette("total-spending", package = "uscogdata")` for the full
|
||||
explanation with worked examples.
|
||||
|
||||
### General vs Total revenue
|
||||
## General vs Total revenue
|
||||
|
||||
`cog_revenue(..., revenue_concept = c("general", "total"))`:
|
||||
`cog_revenue(..., revenue_concept = c("general", "total"))` selects between
|
||||
Census's two published revenue concepts, again defined as crosswalk
|
||||
`revenue_subtype` sets rather than item-code prefixes:
|
||||
|
||||
- **`"general"`** (default) — Census General Revenue: own-source taxes,
|
||||
charges and miscellaneous, plus federal, state and local aid.
|
||||
- **`"total"`** — General plus utility revenue (`A91`–`A94`), liquor store
|
||||
revenue (`A90`), and insurance trust revenue.
|
||||
- `"general"` (the default) is Census **General Revenue**: own-source
|
||||
(taxes, charges, miscellaneous) plus federal, state and local
|
||||
intergovernmental aid.
|
||||
- `"total"` is Census **Total Revenue**: `general` plus utility revenue
|
||||
(`A91`–`A94`), liquor store revenue (`A90`), and insurance trust revenue
|
||||
(unemployment and workers' compensation `Y` codes plus the
|
||||
employee-retirement `X` codes).
|
||||
|
||||
Census defines these by its own identity:
|
||||
The manual defines the first by subtracting the other three from the second,
|
||||
so the two are related by Census's own identity:
|
||||
|
||||
```
|
||||
Total Revenue = General + Utility + Liquor Store + Insurance Trust
|
||||
```
|
||||
|
||||
Two things to know before switching to `"total"`. **Utility revenue is large
|
||||
for cities** — measured on the bundled fixture, utility plus liquor store is
|
||||
15.9% of city revenue, against 1.2% for states and 1.7% for counties. And the
|
||||
**employee-retirement (`X`) codes stop at FY2016**, when those systems moved to
|
||||
the separate Annual Survey of Public Pensions, so a `"total"` series steps down
|
||||
at the FY2016/FY2017 boundary for reasons of collection scope, not revenue
|
||||
(series breaks `SB197`–`SB209`).
|
||||
Two things worth knowing before switching to `"total"`:
|
||||
|
||||
### Reporting coverage: the Census is only sometimes a census
|
||||
- **Utility revenue is large for cities.** Measured on the bundled fixture,
|
||||
utility plus liquor store revenue is 15.9% of city (type 2) revenue, versus
|
||||
1.2% for states and 1.7% for counties. `general` excludes it by definition.
|
||||
- **The employee-retirement (`X`) codes stop at FY2016**, when those systems
|
||||
moved out of the annual finance file into the separate Annual Survey of
|
||||
Public Pensions. A `"total"` series therefore steps down at the
|
||||
FY2016/FY2017 seam for reasons of collection scope, not revenue (series
|
||||
breaks `SB197`–`SB202`, in the corpus's `series_breaks` table).
|
||||
|
||||
**The Census of Governments is a complete enumeration only in years ending in
|
||||
2 and 7.** Every other year is a sample, and the sample varies enormously —
|
||||
measured on the bundled fixture, Wisconsin's 608-city universe rolls up 597
|
||||
governments in FY2012 and 112 in FY2019.
|
||||
## Developer notes
|
||||
|
||||
A statewide total resting on a fifth of the universe looks exactly like one
|
||||
resting on all of it, so every multi-government result now says which it is:
|
||||
### Testing
|
||||
|
||||
The package ships a bundled fixture corpus at `inst/extdata/fixture_corpus/` —
|
||||
a 15 MB four-year slice (2011, 2012, 2019, 2020) of the full corpus covering
|
||||
all 50 states. `tests/testthat/setup.R` automatically points `USCOGDATA_URL`
|
||||
at this fixture, so the full test suite runs offline with no network
|
||||
dependency:
|
||||
|
||||
```r
|
||||
attr(rollup, "provenance")$coverage # per-year n_units_reporting, is_census_year
|
||||
devtools::test() # uses bundled fixture, no credentials required
|
||||
```
|
||||
|
||||
`cog_geographic_rollup()`, `cog_peer_compare()` and `cog_find_peers()` take a
|
||||
`coverage` argument — `"all"` (default), `"census"` (census years only), or
|
||||
`"consistent"` (only units reporting in every requested year, a balanced
|
||||
panel).
|
||||
### Releasing against the live corpus
|
||||
|
||||
`n_units_reporting` is **category-conditional**, and it is not a response rate. A government that was surveyed and genuinely spends
|
||||
nothing in the requested category is indistinguishable from one never surveyed.
|
||||
|
||||
### Absent cells mean two different things
|
||||
|
||||
Before FY2012, an absent cell means Census published `$0`. From FY2012 on, it
|
||||
means not reported. `cog_spending(..., complete = TRUE)` fills the requested
|
||||
grid and labels every row with which it is, via `value_source`:
|
||||
|
||||
| `value_source` | meaning | `amt_nominal` |
|
||||
|---|---|---|
|
||||
| `reported` | the corpus carries this cell | as published |
|
||||
| `census_zero` | dense-source year (≤ FY2011), absent — Census published `$0` | `0` |
|
||||
| `not_reported` | sparse-source year (≥ FY2012), absent — unknown | `NA` |
|
||||
|
||||
That `NA` is deliberate. Filling a modern absence with `0` would invent data.
|
||||
|
||||
### Series breaks surface on their own
|
||||
|
||||
Catalogued breaks that intersect your query appear in provenance whether or not
|
||||
you went looking for them — `series_break_refs` for breaks in a specific item code, and
|
||||
`corpus_break_refs` for caveats about the corpus as a whole (dollar precision
|
||||
across the 1976/1977 boundary, the FY2017 identifier change, the FY2012
|
||||
dense→sparse representation change). `cog_explain()` prints both.
|
||||
|
||||
## How to cite
|
||||
Before cutting a release, run the test suite against the published corpus to
|
||||
catch any drift between the fixture and the real data:
|
||||
|
||||
```r
|
||||
citation("uscogdata")
|
||||
Sys.setenv(USCOGDATA_URL = "<published-corpus-url-with-trailing-slash>")
|
||||
devtools::test()
|
||||
```
|
||||
|
||||
The corpus itself is published under CC-BY-4.0. Cite it as:
|
||||
When the live-corpus run is clean, strip the fixture from the built package by
|
||||
adding this line to `.Rbuildignore`:
|
||||
|
||||
> Civilytics Consulting. US Census of Governments finance corpus.
|
||||
> https://huggingface.co/datasets/civilytics/us-cog-finance
|
||||
```
|
||||
^inst/extdata/fixture_corpus$
|
||||
```
|
||||
|
||||
## Contributing
|
||||
|
||||
Development happens on [Gitea](https://gitea.civilytics.org/Civilytics/uscogdata);
|
||||
[GitHub](https://github.com/civilytics/uscogdata) is a mirror that accepts
|
||||
issues and pull requests. See [CONTRIBUTING.md](CONTRIBUTING.md) for how a
|
||||
patch gets from there to here.
|
||||
|
||||
## License
|
||||
|
||||
MIT © Civilytics Consulting LLC. See [LICENSE.md](LICENSE.md).
|
||||
The test suite is URL-agnostic — `setup.R` falls back to `USCOGDATA_URL` when
|
||||
the bundled fixture is absent, so no test code changes are needed for the
|
||||
release run or after stripping the fixture.
|
||||
|
||||
+5
-21
@@ -1,5 +1,4 @@
|
||||
url: https://civilytics.r-universe.dev/uscogdata
|
||||
|
||||
url: ~
|
||||
template:
|
||||
bootstrap: 5
|
||||
|
||||
@@ -16,26 +15,11 @@ reference:
|
||||
- cog_gov_search
|
||||
- cog_basket_resolution
|
||||
- cog_basket_unresolved
|
||||
- title: Comparison & aggregation
|
||||
desc: Peer cohorts and geographic aggregates.
|
||||
- title: Session
|
||||
contents:
|
||||
- cog_find_peers
|
||||
- cog_peer_compare
|
||||
- cog_geographic_rollup
|
||||
- title: Corpus metadata
|
||||
desc: >
|
||||
What the corpus contains, where a given result came from, and how to
|
||||
hold a local copy of it.
|
||||
contents:
|
||||
- cog_categories
|
||||
- cog_recipes
|
||||
- cog_manifest
|
||||
- cog_explain
|
||||
- cog_mirror
|
||||
- has_keyword("internal")
|
||||
|
||||
articles:
|
||||
- title: Concepts
|
||||
- title: Getting started
|
||||
navbar: ~
|
||||
contents:
|
||||
- total-spending
|
||||
- population-denominators
|
||||
contents: []
|
||||
|
||||
Binary file not shown.
+3
-3
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"schema_version": 6,
|
||||
"built_at": "2026-08-03T16:51:32Z",
|
||||
"pipeline_commit": "e7394a4",
|
||||
"built_at": "2026-07-31T00:47:27Z",
|
||||
"pipeline_commit": "aadb46b",
|
||||
"fixture_note": "Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated from the sparsified schema-v6 corpus: the wide era (<= FY2011) no longer stores explicit zeros, so FY2011 absence means Census published $0 while FY2012+ absence means not reported. representation.parquet and code_set.parquet carry that rule and ship in full, as do every other metadata table in the publish tree. 2011/2012 straddle both the wide-aggregate -> modern-leaf format boundary (exercised by basis=\"harmonized\" and recipe= queries) and the dense -> sparse representation boundary (SB194); 2019/2020 retain the prior per-capita/CPI regression anchors. Regenerated via data-raw/regenerate_fixture_corpus.R.",
|
||||
"data_vintage": {
|
||||
"source_vintages": {
|
||||
@@ -105,7 +105,7 @@
|
||||
},
|
||||
{
|
||||
"path": "data/series_breaks.parquet",
|
||||
"sha256": "731998516cd802f63fcf7fb66053c7a62b7be955ab0794cad4a4979cb7628b87",
|
||||
"sha256": "06dcc995ff533e57cc65fa25086cc9bf83ba592c58bf7cc99269dc2576f69944",
|
||||
"description": "series_breaks.parquet"
|
||||
},
|
||||
{
|
||||
|
||||
@@ -22,8 +22,8 @@
|
||||
"description": "How the intergovernmental leg was assembled; null for 'primary' and 'direct'."
|
||||
},
|
||||
"expenditure_concept_direct_suppressed": {
|
||||
"type": ["boolean", "null"],
|
||||
"description": "TRUE when expenditure_concept = 'total' and at least one requested (year, category) has intergovernmental rows but NO Direct rows in this corpus (typically a legacy aggregate-only family) -- those result rows report the intergovernmental leg alone, not Direct + IG. Always FALSE for expenditure_concept = 'primary' or 'direct'. null (NA) when expenditure_concept = 'total' AND category = 'All Categories': the detector keys on per-category rows, which that mode collapses, so suppression cannot be computed -- see `expenditure_concept_note`. See the affected rows' `notes` for the recovering recipe, if any."
|
||||
"type": "boolean",
|
||||
"description": "TRUE when expenditure_concept = 'total' and at least one requested (year, category) has intergovernmental rows but NO Direct rows in this corpus (typically a legacy aggregate-only family) -- those result rows report the intergovernmental leg alone, not Direct + IG. Always FALSE for expenditure_concept = 'primary' or 'direct'. See the affected rows' `notes` for the recovering recipe, if any."
|
||||
},
|
||||
"revenue_concept": {
|
||||
"type": "string",
|
||||
@@ -33,48 +33,7 @@
|
||||
},
|
||||
"harmonization": { "type": "object" },
|
||||
"recipe": { "type": ["object", "null"] },
|
||||
"suggestions": {
|
||||
"type": "array",
|
||||
"description": "Harmonization recipes that would fill incomplete coverage in the requested years for this government. Empty on a healthy query, on an un-scoped (category = NULL) query, on basis = 'raw', and on a recipe = query (which resolves its own coverage).",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"required": ["recipe_id", "label", "available_years", "hint", "ig_recipe_id",
|
||||
"trigger", "suppressed_amount", "suppressed_years", "suppressed_codes"],
|
||||
"properties": {
|
||||
"recipe_id": { "type": "string" },
|
||||
"label": { "type": "string" },
|
||||
"available_years": {
|
||||
"type": "array",
|
||||
"items": { "type": "integer" },
|
||||
"description": "[year_min, year_max] of the recipe's component coverage."
|
||||
},
|
||||
"hint": { "type": "string" },
|
||||
"ig_recipe_id": {
|
||||
"type": ["string", "null"],
|
||||
"description": "The intergovernmental (M/L) counterpart recipe covering the same function suffixes, or null. Never set for revenue recipes."
|
||||
},
|
||||
"trigger": {
|
||||
"type": "string",
|
||||
"enum": ["empty_year", "suppressed_component"],
|
||||
"description": "Why this fired. 'empty_year': the result has no rows at all in a requested year. 'suppressed_component': the result HAS rows, but a component code carries dollars this government reports in the requested years that the verb's underlying long view structurally excludes -- aggregate-published, carrying no harmonized code, or absent from summary_categories. This is NOT the same thing as 'excluded from the result': a component present in the view under a different category (a scoping choice, e.g. a different `category` or a narrower `expenditure_concept`) contributes 0 and never fires. 'empty_year' wins when both apply, being the stronger claim; the suppressed_* fields are populated either way, using the same underlying-view measurement, and can be 0 even on an 'empty_year' fire."
|
||||
},
|
||||
"suppressed_amount": {
|
||||
"type": "number",
|
||||
"description": "Full US dollars this government reports, in the recipe's component codes, in the requested years, that the verb's underlying long view structurally excludes (aggregate-published, carrying no harmonized code, or absent from summary_categories) -- summed across those years. This is NOT the same quantity as 'what the result excludes': a component present in the view under a different category or a narrower `expenditure_concept` is scoped out on purpose, counts as 0 here, and is not suppression. 0 does not always mean full coverage -- see 'trigger' and 'empty_year'. May be negative where Census publishes a negative `amt` for the excluded rows."
|
||||
},
|
||||
"suppressed_years": {
|
||||
"type": "array",
|
||||
"items": { "type": "integer" },
|
||||
"description": "The requested years contributing to suppressed_amount."
|
||||
},
|
||||
"suppressed_codes": {
|
||||
"type": "array",
|
||||
"items": { "type": "string" },
|
||||
"description": "The excluded component item codes, sorted."
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"suggestions": { "type": "array" },
|
||||
"scope": { "type": "object" },
|
||||
"codes_summed": { "type": "object" },
|
||||
"aggregate_fallback": { "type": ["object", "null"] },
|
||||
|
||||
@@ -1,7 +1,3 @@
|
||||
CREATE OR REPLACE VIEW long AS
|
||||
SELECT *
|
||||
-- {long_files} carries its own quoting: a bracketed list of every partition
|
||||
-- the manifest enumerates, or a single quoted glob on fallback. Do NOT wrap
|
||||
-- it in quotes. See .long_files_sql() in R/views.R for why a glob alone
|
||||
-- cannot work over HTTP.
|
||||
FROM read_parquet({long_files}, hive_partitioning = true);
|
||||
FROM read_parquet('{url}data/long/**/*.parquet', hive_partitioning = true);
|
||||
|
||||
+4
-50
@@ -5,23 +5,18 @@
|
||||
\title{Cash and security holdings for one or more governments}
|
||||
\usage{
|
||||
cog_balances(
|
||||
govid = NULL,
|
||||
govid,
|
||||
years,
|
||||
category = NULL,
|
||||
per_capita = FALSE,
|
||||
adjust_to_year = NULL,
|
||||
basis = c("harmonized", "raw"),
|
||||
recipe = NULL,
|
||||
state = NULL,
|
||||
type = NULL,
|
||||
limit = NULL,
|
||||
offset = NULL
|
||||
recipe = NULL
|
||||
)
|
||||
}
|
||||
\arguments{
|
||||
\item{govid}{Canonical govid(s): a character vector, or a data frame with a
|
||||
`canonical_govid` column (e.g. from [cog_gov_search()]). `NULL` to name
|
||||
the cohort by `state`/`type` instead.}
|
||||
`canonical_govid` column (e.g. from [cog_gov_search()]).}
|
||||
|
||||
\item{years}{Integer vector of fiscal years.}
|
||||
|
||||
@@ -33,12 +28,7 @@ argument: for holdings, `category` is a strict coarsening of
|
||||
every combination would be either redundant or empty.
|
||||
`category = "Fund Balances"` is exactly the `general` family
|
||||
(`W01`/`W31`/`W61`). `balance_subtype` is returned, so a finer split is
|
||||
one `dplyr::filter()` away. The reserved pseudo-category
|
||||
`"All Categories"` (see [cog_spending()]) is **not** supported here and
|
||||
errors with class `uscogdata_all_categories_unsupported`: it sums a
|
||||
concept's subtype scope, and holdings are a stock with no concept
|
||||
vocabulary to sum across. Omit `category` to get every category broken
|
||||
out instead.}
|
||||
one `dplyr::filter()` away.}
|
||||
|
||||
\item{per_capita}{Divide holdings by population. Note this is a **stock per
|
||||
resident** (reserves per person), which is *not* comparable to
|
||||
@@ -54,38 +44,6 @@ and raw space are identical for holdings. Reported in
|
||||
\item{recipe}{Optional harmonization recipe id (see [cog_recipes()]).
|
||||
`"cash_securities_z77_wide"` and `"cash_securities_z78_wide"` bridge the
|
||||
wide era to the modern one.}
|
||||
|
||||
\item{state, type}{Name the cohort by predicate instead of by id: `state` is
|
||||
a 2-letter USPS abbreviation (or a FIPS code) and `type` is one of
|
||||
`"state"`, `"county"`, `"city"`, `"township"` (or the integer `0:3`) --
|
||||
the same vocabulary, and the same internal coercion, as
|
||||
[cog_gov_search()]. Both default to `NULL`.
|
||||
|
||||
The cohort is then expressed as a subquery against `canonical_fips_xwalk`
|
||||
inside each statement rather than round-tripped through R as a literal id
|
||||
list. For a fleet-scale cohort that is the difference between a
|
||||
301,591-character `IN` list re-parsed in 5--8 statements per call and a
|
||||
constant-size predicate: measured at **94 ms versus 449 ms** for the same
|
||||
FY2022 aggregate over the 20,106-government `type = "city"` cohort, within
|
||||
7% of the no-filter floor.
|
||||
|
||||
Supplying `govid` **and** `state`/`type` INTERSECTS them -- the
|
||||
governments in `govid` that also match the predicate -- rather than one
|
||||
silently taking precedence. Naming no cohort at all (`govid`, `state` and
|
||||
`type` all `NULL`) aborts with class `uscogdata_no_cohort`.
|
||||
|
||||
When the cohort is named by predicate, `provenance$scope$govids_found`
|
||||
and `govids_missing` are empty -- there is no id list to report against --
|
||||
and `provenance$scope$cohort` carries `state`, `type` and
|
||||
`n_governments` instead. A `govid`-named cohort reports exactly as before.}
|
||||
|
||||
\item{limit}{Maximum number of result rows to return, pushed into the SQL
|
||||
rather than applied after materializing every row. `NULL` (default)
|
||||
returns everything. Cannot be combined with `recipe` -- see `offset` and
|
||||
`total_rows`.}
|
||||
|
||||
\item{offset}{Rows to skip before `limit` starts counting (0-based).
|
||||
Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set.}
|
||||
}
|
||||
\value{
|
||||
Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||
@@ -104,10 +62,6 @@ Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||
and `truncated` (the observed subtypes whose coverage falls short of the
|
||||
requested years). `expenditure_concept`/`revenue_concept` are `NA` --
|
||||
holdings are a stock, not a flow, so neither concept vocabulary applies.
|
||||
|
||||
When `limit` is set, also carries a `total_rows` attribute: the full
|
||||
unpaginated row count, computed by the same query (`COUNT(*) OVER()`)
|
||||
rather than a second scan.
|
||||
}
|
||||
\description{
|
||||
Returns Census cash-and-security holdings (`category_type = "balance"`):
|
||||
|
||||
@@ -16,10 +16,7 @@ balance), `"spending"`, `"revenue"`, or `"balance"`.}
|
||||
\value{
|
||||
Tibble with columns `category`, `category_type`, `subtype`,
|
||||
`n_codes`, `item_codes` (comma-separated, alphabetical). Sorted by
|
||||
`category_type`, `category`, `subtype`. Includes one row per flow for the
|
||||
reserved pseudo-category `"All Categories"`, which carries `NA` for
|
||||
`subtype`, `n_codes` and `item_codes` because it is a query mode rather
|
||||
than a crosswalk entry — see [cog_spending()]'s `category` argument.
|
||||
`category_type`, `category`, `subtype`.
|
||||
}
|
||||
\description{
|
||||
Returns the category taxonomy exposed by the corpus's
|
||||
|
||||
@@ -21,33 +21,3 @@ Prints the structured provenance attached to a tibble returned by any
|
||||
`cog_*` verb, or returns it as a list for downstream use (MCP tools,
|
||||
dashboards, JSON export).
|
||||
}
|
||||
\section{Two kinds of series break}{
|
||||
|
||||
Catalogued breaks reach you without being asked for, in two disjoint
|
||||
fields, because a caveat about one series and a caveat about the whole
|
||||
corpus are different claims:
|
||||
|
||||
* **`series_break_refs`** — breaks matched against the item codes actually
|
||||
present in this result. A break in one code you queried.
|
||||
* **`corpus_break_refs`** — breaks catalogued with `fin_code = "ALL"`,
|
||||
which are statements about the corpus rather than about any one code:
|
||||
dollar precision across the 1976/1977 boundary (`SB085`), imputation
|
||||
exclusion from FY2002 (`SB087`), the FY2012 dense-to-sparse
|
||||
representation change (`SB194`), and the FY2017 government-identifier
|
||||
change (`SB086`). These are selected on the break-year window alone.
|
||||
|
||||
`SB194` is the one most likely to matter: a query spanning FY2011 to FY2012
|
||||
crosses the boundary where an absent cell stops meaning "Census published
|
||||
$0" and starts meaning "not reported".
|
||||
}
|
||||
|
||||
\section{Other provenance blocks}{
|
||||
|
||||
`transformations$units_conversion` records the `$1,000s`-to-dollars
|
||||
multiply that every amount column has already had applied.
|
||||
`transformations$per_capita` records the population denominator and its
|
||||
year range. `coverage` and `coverage_mode` appear on multi-government
|
||||
results (see [cog_geographic_rollup()]). `completion` appears when
|
||||
`complete = TRUE`. `balance_caveats` appears on [cog_balances()] results.
|
||||
}
|
||||
|
||||
|
||||
@@ -20,11 +20,7 @@ cog_geographic_rollup(
|
||||
`canonical_govid` values. At least one layer required.}
|
||||
|
||||
\item{category}{Single category name or character vector (passed through
|
||||
to [cog_spending()]), or the reserved `"All Categories"` for one summed
|
||||
row per `(year, canonical_govid, subtype)` covering every category in the
|
||||
concept's scope. `"All Categories"` is the efficient way to build a
|
||||
geographic total: without it a caller must issue one rollup per category
|
||||
and sum the results themselves.}
|
||||
to [cog_spending()]).}
|
||||
|
||||
\item{years}{Integer vector of years.}
|
||||
|
||||
@@ -84,23 +80,3 @@ the result. The dropped govids are recorded in
|
||||
(gov type 4) and school districts (gov type 5) from per-capita rollups
|
||||
by design — see `vignette('population-denominators')`.
|
||||
}
|
||||
\section{Reading `coverage`}{
|
||||
|
||||
`provenance$coverage` reports `n_units_reporting` against
|
||||
`n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
||||
it counts governments with rows for the category you asked for, not
|
||||
governments collected that year.** A government that was surveyed and
|
||||
genuinely spends nothing in that category is indistinguishable here from one
|
||||
that was never surveyed.
|
||||
|
||||
The ratio is therefore **not a response rate** and must not be used as one.
|
||||
In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||
`category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||
contract policing to the county sheriff, not non-response.
|
||||
|
||||
The comparison that *is* valid is the same category across a census year
|
||||
(ending in 2 or 7) and a sample year, where the real-zero component is
|
||||
roughly constant and the difference reflects the survey cycle. `is_census_year`
|
||||
marks which is which.
|
||||
}
|
||||
|
||||
|
||||
+4
-24
@@ -4,13 +4,7 @@
|
||||
\alias{cog_gov_search}
|
||||
\title{Search for governments by name, state, and/or type}
|
||||
\usage{
|
||||
cog_gov_search(
|
||||
name = NULL,
|
||||
state = NULL,
|
||||
type = NULL,
|
||||
limit = NULL,
|
||||
offset = NULL
|
||||
)
|
||||
cog_gov_search(name = NULL, state = NULL, type = NULL)
|
||||
}
|
||||
\arguments{
|
||||
\item{name}{Character vector of place name(s). Length 1 = utility mode;
|
||||
@@ -25,26 +19,12 @@ all entries; otherwise must match `length(name)`.}
|
||||
in basket mode (recycles from length 1). Excluded types `4`/`5` (or
|
||||
`"special_district"` / `"school_district"`) trigger an explanatory
|
||||
message and an empty result.}
|
||||
|
||||
\item{limit}{Maximum number of rows to return, applied in SQL. `NULL`
|
||||
(default) returns every match -- which, with no other filter, is the
|
||||
entire crosswalk. Utility mode only: pagination has no meaning in basket
|
||||
mode, where the result is one resolved row per requested name in input
|
||||
order, and is refused there with class
|
||||
`uscogdata_basket_pagination_conflict`.}
|
||||
|
||||
\item{offset}{Rows to skip before `limit` starts counting (0-based).
|
||||
Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set.}
|
||||
}
|
||||
\value{
|
||||
A tibble of `canonical_fips_xwalk` rows. In utility mode, all
|
||||
matches sorted by `population_acs` desc, ties broken by
|
||||
`canonical_govid`. In basket mode, resolved rows in input order, with
|
||||
`attr(., "resolution")` set to the sidecar tibble.
|
||||
|
||||
When `limit` is set, carries a `total_rows` attribute: the full
|
||||
unpaginated match count, computed by the same query (`COUNT(*) OVER()`)
|
||||
rather than a second scan.
|
||||
matches sorted by `population_acs` desc. In basket mode, resolved
|
||||
rows in input order, with `attr(., "resolution")` set to the
|
||||
sidecar tibble.
|
||||
}
|
||||
\description{
|
||||
Resolves human-readable place names into rows of `canonical_fips_xwalk`,
|
||||
|
||||
@@ -107,23 +107,3 @@ call. Those summary rows are quantiles **within each category**, not
|
||||
quantiles of each peer's total — see the `@return` section before summing
|
||||
them.
|
||||
}
|
||||
\section{Reading `coverage`}{
|
||||
|
||||
`provenance$coverage` reports `n_units_reporting` against
|
||||
`n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
||||
it counts cohort members with rows for the category you asked for, not
|
||||
cohort members collected that year.** A government that was surveyed and
|
||||
genuinely spends nothing in that category is indistinguishable here from one
|
||||
that was never surveyed.
|
||||
|
||||
The ratio is therefore **not a response rate** and must not be used as one.
|
||||
In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||
`category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||
contract policing to the county sheriff, not non-response.
|
||||
|
||||
The comparison that *is* valid is the same category across a census year
|
||||
(ending in 2 or 7) and a sample year, where the real-zero component is
|
||||
roughly constant and the difference reflects the survey cycle. `is_census_year`
|
||||
marks which is which.
|
||||
}
|
||||
|
||||
|
||||
+4
-52
@@ -5,7 +5,7 @@
|
||||
\title{Summarized revenue by category}
|
||||
\usage{
|
||||
cog_revenue(
|
||||
govid = NULL,
|
||||
govid,
|
||||
years,
|
||||
category = NULL,
|
||||
per_capita = FALSE,
|
||||
@@ -13,31 +13,16 @@ cog_revenue(
|
||||
basis = c("harmonized", "raw"),
|
||||
recipe = NULL,
|
||||
revenue_concept = c("general", "total"),
|
||||
complete = FALSE,
|
||||
limit = NULL,
|
||||
offset = NULL,
|
||||
state = NULL,
|
||||
type = NULL
|
||||
complete = FALSE
|
||||
)
|
||||
}
|
||||
\arguments{
|
||||
\item{govid}{Character vector of `canonical_govid` values, or `NULL` to name
|
||||
the cohort by `state`/`type` instead. One of `govid`, `state`, or `type`
|
||||
is required.}
|
||||
\item{govid}{Character vector of `canonical_govid` values.}
|
||||
|
||||
\item{years}{Integer vector of years.}
|
||||
|
||||
\item{category}{Character vector of category names (from
|
||||
`summary_categories.category`), or `NULL` for all categories broken out
|
||||
one row each. The reserved value `"All Categories"` instead returns a
|
||||
single summed row per `(year, canonical_govid, subtype)`, covering every
|
||||
category inside the requested concept's subtype scope. It cannot be
|
||||
combined with other category names, and it is not the same thing as
|
||||
`revenue_concept = "total"`: the concept chooses which subtypes are in
|
||||
scope, `"All Categories"` chooses whether rows inside that scope are
|
||||
broken out or summed. Because the result keeps one row per
|
||||
`revenue_subtype`, filtering the returned frame to
|
||||
`revenue_subtype == "own_source"` gives an own-source revenue total.}
|
||||
`summary_categories.category`), or `NULL` for all categories.}
|
||||
|
||||
\item{per_capita}{If `TRUE`, adds `amt_per_capita_nominal` (and
|
||||
`amt_per_capita_real` when `adjust_to_year` is set) using the per-year
|
||||
@@ -121,39 +106,6 @@ possibly-misleading `"harmonized"`/`"raw"` value.}
|
||||
`recipe` or with `expenditure_concept = "total"` (class
|
||||
`uscogdata_complete_unsupported`) — neither draws its cells from
|
||||
`code_set`.}
|
||||
|
||||
\item{limit}{Maximum number of result rows to return, pushed into the SQL
|
||||
query itself (`LIMIT`/`OFFSET`) rather than applied after the full
|
||||
result is materialized. `NULL` (the default) returns every matching row,
|
||||
exactly as before this parameter existed. Mutually exclusive with
|
||||
`recipe` and with `complete = TRUE` -- see `offset` and `total_rows`.}
|
||||
|
||||
\item{offset}{Rows to skip before `limit` starts counting (0-based).
|
||||
Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set.}
|
||||
|
||||
\item{state, type}{Name the cohort by predicate instead of by id: `state` is
|
||||
a 2-letter USPS abbreviation (or a FIPS code) and `type` is one of
|
||||
`"state"`, `"county"`, `"city"`, `"township"` (or the integer `0:3`) --
|
||||
the same vocabulary, and the same internal coercion, as
|
||||
[cog_gov_search()]. Both default to `NULL`.
|
||||
|
||||
The cohort is then expressed as a subquery against `canonical_fips_xwalk`
|
||||
inside each statement rather than round-tripped through R as a literal id
|
||||
list. For a fleet-scale cohort that is the difference between a
|
||||
301,591-character `IN` list re-parsed in 5--8 statements per call and a
|
||||
constant-size predicate: measured at **94 ms versus 449 ms** for the same
|
||||
FY2022 aggregate over the 20,106-government `type = "city"` cohort, within
|
||||
7% of the no-filter floor.
|
||||
|
||||
Supplying `govid` **and** `state`/`type` INTERSECTS them -- the
|
||||
governments in `govid` that also match the predicate -- rather than one
|
||||
silently taking precedence. Naming no cohort at all (`govid`, `state` and
|
||||
`type` all `NULL`) aborts with class `uscogdata_no_cohort`.
|
||||
|
||||
When the cohort is named by predicate, `provenance$scope$govids_found`
|
||||
and `govids_missing` are empty -- there is no id list to report against --
|
||||
and `provenance$scope$cohort` carries `state`, `type` and
|
||||
`n_governments` instead. A `govid`-named cohort reports exactly as before.}
|
||||
}
|
||||
\value{
|
||||
Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||
|
||||
+6
-63
@@ -5,7 +5,7 @@
|
||||
\title{Summarized spending by category}
|
||||
\usage{
|
||||
cog_spending(
|
||||
govid = NULL,
|
||||
govid,
|
||||
years,
|
||||
category = NULL,
|
||||
per_capita = FALSE,
|
||||
@@ -13,31 +13,16 @@ cog_spending(
|
||||
basis = c("harmonized", "raw"),
|
||||
recipe = NULL,
|
||||
expenditure_concept = c("primary", "direct", "total"),
|
||||
complete = FALSE,
|
||||
limit = NULL,
|
||||
offset = NULL,
|
||||
state = NULL,
|
||||
type = NULL
|
||||
complete = FALSE
|
||||
)
|
||||
}
|
||||
\arguments{
|
||||
\item{govid}{Character vector of `canonical_govid` values, or `NULL` to name
|
||||
the cohort by `state`/`type` instead. One of `govid`, `state`, or `type`
|
||||
is required.}
|
||||
\item{govid}{Character vector of `canonical_govid` values.}
|
||||
|
||||
\item{years}{Integer vector of years.}
|
||||
|
||||
\item{category}{Character vector of category names (from
|
||||
`summary_categories.category`), or `NULL` for all categories broken out
|
||||
one row each. The reserved value `"All Categories"` instead returns a
|
||||
single summed row per `(year, canonical_govid, subtype)`, covering every
|
||||
category inside the requested concept's subtype scope. It cannot be
|
||||
combined with other category names, and it is not the same thing as
|
||||
`expenditure_concept = "total"`: the concept chooses which subtypes are in
|
||||
scope, `"All Categories"` chooses whether rows inside that scope are
|
||||
broken out or summed. Because the result keeps one row per
|
||||
`spend_subtype`, filtering the returned frame to
|
||||
`spend_subtype == "operations"` gives an operating-expenditure total.}
|
||||
`summary_categories.category`), or `NULL` for all categories.}
|
||||
|
||||
\item{per_capita}{If `TRUE`, adds `amt_per_capita_nominal` (and
|
||||
`amt_per_capita_real` when `adjust_to_year` is set) using the per-year
|
||||
@@ -112,12 +97,7 @@ possibly-misleading `"harmonized"`/`"raw"` value.}
|
||||
component (when one exists), and
|
||||
`provenance$expenditure_concept_direct_suppressed` is `TRUE` -- the
|
||||
figure in those rows is the intergovernmental leg alone, not Direct +
|
||||
IG. When `category = "All Categories"` is combined with
|
||||
`expenditure_concept = "total"`, this detection cannot run (it keys on
|
||||
per-category rows, which all-categories mode collapses to one literal
|
||||
value), so `expenditure_concept_direct_suppressed` is `NA` rather than a
|
||||
possibly-false `FALSE`; query an explicit `category` to get a real
|
||||
answer.}
|
||||
IG.}
|
||||
|
||||
\item{complete}{If `TRUE`, fill the requested grid so that a cell the
|
||||
corpus does not carry still appears, labelled with **why** it is
|
||||
@@ -141,39 +121,6 @@ possibly-misleading `"harmonized"`/`"raw"` value.}
|
||||
`recipe` or with `expenditure_concept = "total"` (class
|
||||
`uscogdata_complete_unsupported`) — neither draws its cells from
|
||||
`code_set`.}
|
||||
|
||||
\item{limit}{Maximum number of result rows to return, pushed into the SQL
|
||||
query itself (`LIMIT`/`OFFSET`) rather than applied after the full
|
||||
result is materialized. `NULL` (the default) returns every matching row,
|
||||
exactly as before this parameter existed. Mutually exclusive with
|
||||
`recipe` and with `complete = TRUE` -- see `offset` and `total_rows`.}
|
||||
|
||||
\item{offset}{Rows to skip before `limit` starts counting (0-based).
|
||||
Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set.}
|
||||
|
||||
\item{state, type}{Name the cohort by predicate instead of by id: `state` is
|
||||
a 2-letter USPS abbreviation (or a FIPS code) and `type` is one of
|
||||
`"state"`, `"county"`, `"city"`, `"township"` (or the integer `0:3`) --
|
||||
the same vocabulary, and the same internal coercion, as
|
||||
[cog_gov_search()]. Both default to `NULL`.
|
||||
|
||||
The cohort is then expressed as a subquery against `canonical_fips_xwalk`
|
||||
inside each statement rather than round-tripped through R as a literal id
|
||||
list. For a fleet-scale cohort that is the difference between a
|
||||
301,591-character `IN` list re-parsed in 5--8 statements per call and a
|
||||
constant-size predicate: measured at **94 ms versus 449 ms** for the same
|
||||
FY2022 aggregate over the 20,106-government `type = "city"` cohort, within
|
||||
7% of the no-filter floor.
|
||||
|
||||
Supplying `govid` **and** `state`/`type` INTERSECTS them -- the
|
||||
governments in `govid` that also match the predicate -- rather than one
|
||||
silently taking precedence. Naming no cohort at all (`govid`, `state` and
|
||||
`type` all `NULL`) aborts with class `uscogdata_no_cohort`.
|
||||
|
||||
When the cohort is named by predicate, `provenance$scope$govids_found`
|
||||
and `govids_missing` are empty -- there is no id list to report against --
|
||||
and `provenance$scope$cohort` carries `state`, `type` and
|
||||
`n_governments` instead. A `govid`-named cohort reports exactly as before.}
|
||||
}
|
||||
\value{
|
||||
Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||
@@ -183,11 +130,7 @@ Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||
and `value_source` when `complete = TRUE`.
|
||||
Carries a `provenance` attribute matching `inst/schemas/provenance-v1.json`,
|
||||
whose `completion` block reports `applied`, `rows_filled`, and the
|
||||
per-year `absence_means` rule that was applied. When `limit` is set,
|
||||
also carries a `total_rows` attribute: the full unpaginated row count,
|
||||
computed by the same query (`COUNT(*) OVER()`) rather than a second
|
||||
round trip -- so a caller walking pages never has to ask "how many are
|
||||
there" separately.
|
||||
per-year `absence_means` rule that was applied.
|
||||
}
|
||||
\description{
|
||||
One row per `(year, canonical_govid, spend_subtype, category)`. Amounts are
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,14 +0,0 @@
|
||||
# Project journal
|
||||
|
||||
Append-only, newest first. **Entries are never edited** — the value of this file is
|
||||
that it records what was believed at the time, including the parts that turned out
|
||||
wrong. Where things stand *today* is in `STATUS.md`, which is generated.
|
||||
|
||||
Four lines per entry. The analysis belongs in the issue or the decision record; this
|
||||
file carries the reasoning and the pointers.
|
||||
|
||||
- **Why** — the driver. The one line git cannot reconstruct later.
|
||||
- **Obligates** — issues this change created elsewhere. Numbers, not prose.
|
||||
- **Refs** — commits, issues, decision records.
|
||||
|
||||
---
|
||||
@@ -1,67 +0,0 @@
|
||||
# Project status
|
||||
|
||||
> Between the compass markers is generated. Edit the sources, not this.
|
||||
|
||||
<!-- compass:begin -->
|
||||
<!-- compass:board -->
|
||||
|
||||
## Where this stands
|
||||
|
||||
uscogdata is at 0.4.0 and its public surface is settled: the query verbs, the cohort
|
||||
predicates added in this release, and the provenance contract every verb returns.
|
||||
|
||||
The six open issues split cleanly. Two are API work carried out of the #9 review pass
|
||||
and deliberately deferred there rather than fixed in that branch. Three concern the
|
||||
corpus layer, and the largest of them, partition-level caching, was named the single
|
||||
highest-leverage change on the remote path before being deferred. One, the
|
||||
data-correction intake (#52), is a decision rather than a task: it was parked during
|
||||
the 0.3.0 design, and the API announcement waits on it, because without it the corpus
|
||||
cannot make the "traceable and correctable" claim that most distinguishes it from
|
||||
Census's own files.
|
||||
|
||||
Nothing here is blocked on anything else, so the ordering is a judgement about value
|
||||
rather than a dependency graph.
|
||||
|
||||
Compass's own files moved out of `docs/` this session. They were sitting inside
|
||||
pkgdown's output directory, and `pkgdown::clean_site()` deletes every top-level entry
|
||||
there except `CNAME` and `dev` — asked directly, it listed `docs/pm` and
|
||||
`docs/decisions` among the 28 it would remove, with the guard that would have stopped
|
||||
it satisfied by `docs/pkgdown.yml`. They are in `pm/` now. Nothing was lost: the
|
||||
journal had no entries and there were no decision records yet, which made this the
|
||||
cheapest moment to move. The `.gitignore` workaround that re-included two children of
|
||||
an excluded `docs/` is gone with it.
|
||||
|
||||
## Ready to work on next
|
||||
|
||||
- **#34** cog_revenue() offers expenditure recipes as suggestions: scope the candidate query by category_type · `ws/api` — nothing is blocking it; something is currently wrong
|
||||
- **#36** n_units_reporting is category-conditional and cannot be read as a response rate · `ws/corpus` — nothing is blocking it; owed work from an earlier change
|
||||
- **#2** Extend population data to be households as an alternate spending denominator · `ws/corpus` — nothing is blocking it
|
||||
- **#33** Decompose .build_suggestions() (106 lines) into named helpers · `ws/api` — nothing is blocking it
|
||||
- **#52** Release 11/11: design the data-correction intake (deferred; gates the API announcement) · `ws/corpus` — nothing is blocking it
|
||||
- **#64** Partition-level caching: R/cache.R is still a stub, and the remote path pays for it every session · `ws/corpus` — nothing is blocking it
|
||||
|
||||
## Workstreams
|
||||
|
||||
| Stream | Commits since | Open | Debt | Owes docs |
|
||||
|---|---|---|---|---|
|
||||
| Query verbs and results | 77 | 2 | 0 | no |
|
||||
| Corpus, mirror, provenance | 39 | 4 | 1 | no |
|
||||
| Vignettes and guides | 34 | 0 | 0 | **yes** |
|
||||
|
||||
## CI
|
||||
|
||||

|
||||

|
||||
|
||||
<details>
|
||||
<summary>Dependency graph and detail</summary>
|
||||
|
||||
_Nothing blocks anything else, so there is no graph to draw._
|
||||
|
||||
- Marker: `none` (no journal entry yet)
|
||||
- Commits since: 165
|
||||
- Open issues: 6
|
||||
|
||||
</details>
|
||||
|
||||
<!-- compass:end -->
|
||||
@@ -1,44 +0,0 @@
|
||||
[project]
|
||||
name = "uscogdata"
|
||||
forge = "Civilytics/uscogdata"
|
||||
|
||||
# Three strands that go stale independently: what the verbs return, what the
|
||||
# corpus is and how it is mounted, and how both are explained to a reader.
|
||||
|
||||
[[workstream]]
|
||||
id = "api"
|
||||
title = "Query verbs and results"
|
||||
paths = [
|
||||
"R/revenue.R", "R/spending.R", "R/balances.R", "R/peers.R", "R/search.R",
|
||||
"R/categories.R", "R/recipes.R", "R/rollup.R", "R/explain.R", "R/basket.R",
|
||||
"R/suggestions.R", "R/suppression.R", "R/complete.R", "R/cohort.R",
|
||||
"R/basis.R", "R/adjust.R", "R/pagination.R",
|
||||
]
|
||||
docs = ["vignettes/*.Rmd", "README.md"]
|
||||
|
||||
[[workstream]]
|
||||
id = "corpus"
|
||||
title = "Corpus, mirror, provenance"
|
||||
paths = [
|
||||
"R/manifest.R", "R/mirror.R", "R/cache.R", "R/session.R", "R/provenance.R",
|
||||
"R/coverage.R", "R/config.R", "R/views.R", "R/series_breaks.R",
|
||||
"R/balance_caveats.R", "R/zzz.R", "data-raw/**", "inst/sql/**",
|
||||
]
|
||||
docs = ["vignettes/*.Rmd", "NEWS.md"]
|
||||
|
||||
[[workstream]]
|
||||
id = "docs"
|
||||
title = "Vignettes and guides"
|
||||
paths = ["vignettes/**", "README.md", "_pkgdown.yml", "NEWS.md"]
|
||||
docs = []
|
||||
|
||||
[roborev]
|
||||
project_guidelines = [
|
||||
"Every verb calls .ensure_session() first, then queries via DBI::dbGetQuery().",
|
||||
"A verb's return value is always a tbl_df carrying a provenance attribute.",
|
||||
"govid inputs always go through .coerce_govid_input(); it accepts a character vector or a data frame.",
|
||||
"SQL has two layers: view definitions are numbered .sql files in inst/sql/ registered by .register_views(); query construction is inline sprintf() in R. Add a view as a file; build a query in R.",
|
||||
"No arrow dependency -- DuckDB reads parquet natively.",
|
||||
"withr is Suggests-only and must appear in tests alone.",
|
||||
"Tests must pass offline against the bundled fixture; tests/testthat/setup.R sets USCOGDATA_URL for that.",
|
||||
]
|
||||
@@ -1,12 +0,0 @@
|
||||
# Decisions
|
||||
|
||||
One file per decision, numbered and immutable. A decision that changes is superseded
|
||||
by a new record, never edited in place — the old reasoning is the point.
|
||||
|
||||
The table below is **generated** by `compass:decide`. Do not hand-edit it.
|
||||
|
||||
<!-- compass:begin decisions -->
|
||||
| # | Date | Decision | Status |
|
||||
|---|---|---|---|
|
||||
| — | — | *No decisions recorded yet.* | — |
|
||||
<!-- compass:end decisions -->
|
||||
@@ -1,4 +1,4 @@
|
||||
# `uscogdata` 0.3.0 — public release
|
||||
# `uscogdata` 0.1.0 — public release
|
||||
|
||||
**Date:** 2026-08-08 · **Status:** design, awaiting approval
|
||||
**Scope:** release-readiness, README, NEWS. Distribution mechanics recorded here as
|
||||
@@ -176,25 +176,15 @@ with a URL that resolves for someone who has only this repo.
|
||||
|
||||
## NEWS.md
|
||||
|
||||
`NEWS.md` currently holds two sections. `0.2.0` is a legitimate changelog — the
|
||||
`"All Categories"` reserved value, the coverage-signposting fix, the
|
||||
`n_units_reporting` documentation — and it stays. Beneath it,
|
||||
`0.1.0 (development)` is a pre-release churn log: changes described relative to
|
||||
states no user has ever seen ("Breaking: corpus schema_version 4", "the package
|
||||
now requires…"), spanning the package's entire pre-release development. To a
|
||||
newcomer deciding whether to depend on this, that section reads as instability.
|
||||
The current NEWS is a pre-release churn log: changes described relative to states
|
||||
no user has seen ("Breaking: corpus schema_version 4", "the package now
|
||||
requires…"), newest-first across the package's entire pre-release development
|
||||
(2026-04-23 to 2026-08-04, 140 commits). To a newcomer evaluating whether to
|
||||
depend on the package, it reads as instability.
|
||||
|
||||
**A new `0.3.0` section is added at the top, framed as the first public
|
||||
release**: what the package does, what the corpus covers, and the caveats that
|
||||
are genuinely load-bearing. **`0.2.0` is kept verbatim.** **`0.1.0 (development)`
|
||||
is dropped** — that history stays in git, where it belongs.
|
||||
|
||||
The version is `0.3.0` rather than `0.2.0` because this release changes
|
||||
user-visible behaviour: remote corpus reads go from broken to working, and the
|
||||
default URL from a dead placeholder to a live corpus. It is also not `1.0.0` —
|
||||
the corpus still excludes government types 4 and 5 pending validation, so a
|
||||
stability promise would overclaim. No git tag exists for any prior version;
|
||||
`chore: release 0.2.0` bumped `DESCRIPTION` and `NEWS` only.
|
||||
**0.1.0 is rewritten as an initial release**: what the package does, what the
|
||||
corpus covers, and the caveats that are genuinely load-bearing. The pre-release
|
||||
history is not preserved in NEWS — it is in git, where it belongs.
|
||||
|
||||
The substantive content is migrated, not deleted. These are hard-won and belong
|
||||
in documentation rather than buried in a changelog:
|
||||
@@ -234,7 +224,7 @@ which is a worse first impression than a week's delay.
|
||||
account — r-universe links maintainer identity by matching DESCRIPTION's email
|
||||
against registered GitHub emails, and the association only takes effect on the
|
||||
next build.
|
||||
6. Tag `v0.3.0`. Create `github.com/civilytics/civilytics.r-universe.dev` with a
|
||||
6. Tag `v0.1.0`. Create `github.com/civilytics/civilytics.r-universe.dev` with a
|
||||
`packages.json` pinned to the tag, pointing at the GitHub mirror rather than
|
||||
Gitea so clone traffic stays off maxwell. Install the r-universe app.
|
||||
|
||||
|
||||
@@ -1,58 +0,0 @@
|
||||
test_that('cog_geographic_rollup() accepts "All Categories" and agrees with per-category sums', {
|
||||
skip_if_no_corpus()
|
||||
govs <- cog_gov_search(name = NULL, state = "WI", type = 2L)
|
||||
expect_gt(nrow(govs), 1L)
|
||||
ids <- list(city = utils::head(govs$canonical_govid, 25L))
|
||||
|
||||
by_cat <- cog_geographic_rollup(ids, category = NULL, years = 2019L)
|
||||
total <- cog_geographic_rollup(ids, category = "All Categories", years = 2019L)
|
||||
|
||||
expect_setequal(unique(total$category), "All Categories")
|
||||
# one row per (govid, subtype) that appears in the per-category result
|
||||
key_by_cat <- unique(paste(by_cat$canonical_govid, by_cat$spend_subtype))
|
||||
key_total <- paste(total$canonical_govid, total$spend_subtype)
|
||||
expect_setequal(key_total, key_by_cat)
|
||||
|
||||
lhs <- tapply(by_cat$amt_nominal, paste(by_cat$canonical_govid, by_cat$spend_subtype), sum)
|
||||
rhs <- tapply(total$amt_nominal, key_total, sum)
|
||||
expect_equal(as.numeric(rhs[names(lhs)]), as.numeric(lhs), tolerance = 1e-8)
|
||||
})
|
||||
|
||||
test_that('"All Categories" survives per_capita and inflation adjustment through the rollup', {
|
||||
skip_if_no_corpus()
|
||||
govs <- cog_gov_search(name = NULL, state = "WI", type = 2L)
|
||||
ids <- list(city = utils::head(govs$canonical_govid, 10L))
|
||||
r <- cog_geographic_rollup(ids, category = "All Categories", years = 2019L,
|
||||
per_capita = TRUE, adjust_to_year = 2020L)
|
||||
expect_true(all(c("amt_per_capita_nominal", "amt_real", "amt_per_capita_real") %in% names(r)))
|
||||
expect_setequal(unique(r$category), "All Categories")
|
||||
expect_true(all(is.finite(r$amt_real)))
|
||||
})
|
||||
|
||||
test_that('cog_geographic_rollup() still refuses expenditure_concept = "total" with "All Categories"', {
|
||||
skip_if_no_corpus()
|
||||
govs <- cog_gov_search(name = NULL, state = "WI", type = 2L)
|
||||
ids <- list(city = utils::head(govs$canonical_govid, 5L))
|
||||
expect_error(
|
||||
cog_geographic_rollup(ids, category = "All Categories", years = 2019L,
|
||||
expenditure_concept = "total")
|
||||
)
|
||||
})
|
||||
|
||||
test_that("n_units_reporting is category-conditional, not a response rate", {
|
||||
skip_if_no_corpus()
|
||||
govs <- cog_gov_search(name = NULL, state = "WI", type = 2L)
|
||||
ids <- list(city = govs$canonical_govid)
|
||||
|
||||
police <- cog_geographic_rollup(ids, category = "Police", years = 2012L)
|
||||
allcat <- cog_geographic_rollup(ids, category = "All Categories", years = 2012L)
|
||||
|
||||
cov_police <- cog_explain(police, format = "list")$coverage
|
||||
cov_all <- cog_explain(allcat, format = "list")$coverage
|
||||
|
||||
# Same year, same requested govids, same collection -- yet a single category
|
||||
# reports fewer units than the all-categories query. That gap is real zeros,
|
||||
# not non-response, which is exactly why the ratio is not a response rate.
|
||||
expect_lte(cov_police$n_units_reporting, cov_all$n_units_reporting)
|
||||
expect_identical(cov_police$n_units_expected, cov_all$n_units_expected)
|
||||
})
|
||||
@@ -1,267 +0,0 @@
|
||||
# Baseline at branch point: 843 PASS / 0 FAIL / 0 SKIP / 0 WARN (2026-08-05, origin/main 2fc9e75)
|
||||
|
||||
test_that(".build_verb_sql emits a literal category and no category filter in all-categories mode", {
|
||||
sql <- uscogdata:::.build_verb_sql(
|
||||
view = "spending_annotated",
|
||||
subtype_col = "spend_subtype",
|
||||
cohort = uscogdata:::.make_cohort("552025209777"),
|
||||
years = 2019L,
|
||||
category = NULL,
|
||||
subtype_scope = c("operations", "capital"),
|
||||
all_categories = TRUE
|
||||
)
|
||||
|
||||
expect_match(sql, "'All Categories' AS category", fixed = TRUE)
|
||||
# no category filter of any kind
|
||||
expect_false(grepl("AND category IN", sql, fixed = TRUE))
|
||||
# category is not a grouping key
|
||||
expect_false(grepl("GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, spend_subtype, category",
|
||||
sql, fixed = TRUE))
|
||||
# the subtype allowlist still applies -- this is what makes the sum a concept
|
||||
expect_match(sql, "AND spend_subtype IN ('operations','capital')", fixed = TRUE)
|
||||
})
|
||||
|
||||
test_that(".build_verb_sql is unchanged when all_categories is FALSE", {
|
||||
args <- list(
|
||||
view = "spending_annotated", subtype_col = "spend_subtype",
|
||||
cohort = uscogdata:::.make_cohort("552025209777"), years = 2019L, category = NULL,
|
||||
subtype_scope = c("operations", "capital")
|
||||
)
|
||||
old <- do.call(uscogdata:::.build_verb_sql, args)
|
||||
new <- do.call(uscogdata:::.build_verb_sql, c(args, list(all_categories = FALSE)))
|
||||
expect_identical(old, new)
|
||||
expect_match(new, "GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, spend_subtype, category",
|
||||
fixed = TRUE)
|
||||
})
|
||||
|
||||
test_that(".ALL_CATEGORIES is the exact reserved string", {
|
||||
expect_identical(uscogdata:::.ALL_CATEGORIES, "All Categories")
|
||||
})
|
||||
|
||||
test_that('cog_spending(category = "All Categories") sums to the per-category total', {
|
||||
gov <- "552025209777"
|
||||
by_cat <- cog_spending(gov, 2019L)
|
||||
total <- cog_spending(gov, 2019L, category = "All Categories")
|
||||
|
||||
expect_true(nrow(total) > 0L)
|
||||
expect_setequal(unique(total$category), "All Categories")
|
||||
# one row per subtype present in the by-category result
|
||||
expect_setequal(unique(total$spend_subtype), unique(by_cat$spend_subtype))
|
||||
expect_equal(nrow(total), length(unique(by_cat$spend_subtype)))
|
||||
|
||||
# the dollars agree, per subtype
|
||||
lhs <- tapply(by_cat$amt_nominal, by_cat$spend_subtype, sum)
|
||||
rhs <- tapply(total$amt_nominal, total$spend_subtype, sum)
|
||||
expect_equal(as.numeric(rhs[names(lhs)]), as.numeric(lhs), tolerance = 1e-8)
|
||||
})
|
||||
|
||||
test_that('"All Categories" respects expenditure_concept', {
|
||||
gov <- "552025209777"
|
||||
prim <- cog_spending(gov, 2019L, category = "All Categories",
|
||||
expenditure_concept = "primary")
|
||||
dir <- cog_spending(gov, 2019L, category = "All Categories",
|
||||
expenditure_concept = "direct")
|
||||
# direct = primary plus interest and insurance benefits, so it is never smaller
|
||||
expect_gte(sum(dir$amt_nominal), sum(prim$amt_nominal))
|
||||
})
|
||||
|
||||
test_that('"All Categories" works on revenue and respects revenue_concept', {
|
||||
gov <- "552025209777"
|
||||
gen <- cog_revenue(gov, 2019L, category = "All Categories",
|
||||
revenue_concept = "general")
|
||||
tot <- cog_revenue(gov, 2019L, category = "All Categories",
|
||||
revenue_concept = "total")
|
||||
expect_setequal(unique(gen$category), "All Categories")
|
||||
expect_gte(sum(tot$amt_nominal), sum(gen$amt_nominal))
|
||||
})
|
||||
|
||||
test_that('"All Categories" cannot be combined with another category', {
|
||||
expect_error(
|
||||
cog_spending("552025209777", 2019L, category = c("All Categories", "Police")),
|
||||
class = "uscogdata_all_categories_not_combinable"
|
||||
)
|
||||
})
|
||||
|
||||
test_that('"All Categories" is recorded in provenance', {
|
||||
r <- cog_spending("552025209777", 2019L, category = "All Categories")
|
||||
expect_identical(cog_explain(r, format = "list")$category, "All Categories")
|
||||
})
|
||||
|
||||
test_that('"All Categories" combines with subtype to give operating totals', {
|
||||
gov <- "552025209777"
|
||||
ops_by_cat <- cog_spending(gov, 2019L)
|
||||
ops_by_cat <- ops_by_cat[ops_by_cat$spend_subtype == "operations", ]
|
||||
ops_total <- cog_spending(gov, 2019L, category = "All Categories")
|
||||
ops_total <- ops_total[ops_total$spend_subtype == "operations", ]
|
||||
expect_equal(sum(ops_total$amt_nominal), sum(ops_by_cat$amt_nominal),
|
||||
tolerance = 1e-8)
|
||||
})
|
||||
|
||||
test_that('cog_categories() advertises "All Categories" for both flows', {
|
||||
all <- cog_categories()
|
||||
rows <- all[all$category == "All Categories", ]
|
||||
expect_setequal(rows$category_type, c("expenditure", "revenue"))
|
||||
expect_true(all(is.na(rows$subtype)))
|
||||
expect_true(all(is.na(rows$n_codes)))
|
||||
})
|
||||
|
||||
test_that('cog_categories(type=) still scopes, including the pseudo-category', {
|
||||
sp <- cog_categories(type = "spending")
|
||||
expect_setequal(unique(sp$category_type), "expenditure")
|
||||
expect_true("All Categories" %in% sp$category)
|
||||
|
||||
rev <- cog_categories(type = "revenue")
|
||||
expect_setequal(unique(rev$category_type), "revenue")
|
||||
expect_true("All Categories" %in% rev$category)
|
||||
|
||||
# balances have no concept vocabulary, so no pseudo-category
|
||||
bal <- cog_categories(type = "balance")
|
||||
expect_false("All Categories" %in% bal$category)
|
||||
})
|
||||
|
||||
test_that('cog_categories(pattern=) matches the pseudo-category', {
|
||||
hit <- cog_categories(pattern = "^All Categories$")
|
||||
expect_equal(nrow(hit), 2L)
|
||||
})
|
||||
|
||||
# --- final whole-branch review fixes ---------------------------------------
|
||||
|
||||
test_that('complete = TRUE is refused when combined with "All Categories"', {
|
||||
# .completion_grid_sql() would emit `AND c.category IN ('All Categories')`,
|
||||
# match zero crosswalk rows, and the early return in .complete_result()
|
||||
# would stamp completion$applied = TRUE, rows_filled = 0 -- reading as "the
|
||||
# grid was checked and nothing was missing" when nothing was actually
|
||||
# checked. Filling a summed row has no defined semantics, so the verb must
|
||||
# refuse the combination outright (finding 2).
|
||||
expect_error(
|
||||
cog_spending("552025209777", 2019L, category = "All Categories",
|
||||
complete = TRUE),
|
||||
class = "uscogdata_complete_unsupported"
|
||||
)
|
||||
expect_error(
|
||||
cog_revenue("552025209777", 2019L, category = "All Categories",
|
||||
complete = TRUE),
|
||||
class = "uscogdata_complete_unsupported"
|
||||
)
|
||||
})
|
||||
|
||||
test_that('cog_balances() rejects "All Categories" instead of silently returning zero rows', {
|
||||
# cog_balances() reuses .validate_verb_inputs() but did not pass
|
||||
# allow_all_categories = TRUE, so "All Categories" used to become
|
||||
# `AND category IN ('All Categories')` against balance_annotated -- 0
|
||||
# matching crosswalk rows, 0 rows back, no error (finding 3). Holdings are
|
||||
# a stock with no concept vocabulary to sum across, so the honest answer is
|
||||
# to refuse, the same way cog_spending()/cog_revenue() refuse other
|
||||
# nonsensical combinations.
|
||||
expect_error(
|
||||
cog_balances("552025209777", 2019L, category = "All Categories"),
|
||||
class = "uscogdata_all_categories_unsupported"
|
||||
)
|
||||
# An ordinary category still works -- this is not a blanket regression.
|
||||
r <- suppressMessages(
|
||||
cog_balances("552025209777", 2019L, category = "Fund Balances")
|
||||
)
|
||||
expect_gt(nrow(r), 0L)
|
||||
})
|
||||
|
||||
test_that('expenditure_concept_direct_suppressed is NA, not FALSE, when categories are collapsed', {
|
||||
# .detect_direct_suppressed() keys on
|
||||
# paste(year, canonical_govid, category, sep = "\r"). In all-categories
|
||||
# mode every row carries the literal "All Categories" value, so an IG-only
|
||||
# row's key collides with any ordinary Direct row for the same
|
||||
# (year, govid) -- has_direct reads TRUE whenever the government has ANY
|
||||
# direct spending at all, candidate is always empty, and the detector can
|
||||
# never fire. Before the fix this silently reported FALSE, an affirmative
|
||||
# claim the code did not actually compute (finding 1). NA is the honest
|
||||
# answer: cog_explain(x, format = "list") is required here, since without
|
||||
# format = "list" it returns the result tibble, not the provenance list.
|
||||
gov <- "552025209777"
|
||||
t <- cog_spending(gov, 2019L, category = "All Categories",
|
||||
expenditure_concept = "total")
|
||||
prov <- cog_explain(t, format = "list")
|
||||
expect_true(is.na(prov$expenditure_concept_direct_suppressed))
|
||||
expect_false(isTRUE(prov$expenditure_concept_direct_suppressed))
|
||||
expect_match(prov$expenditure_concept_note, "unavailable", fixed = TRUE)
|
||||
|
||||
# A per-category "total" query on the same government/year is unaffected --
|
||||
# the detector can still key correctly and reports a strict logical.
|
||||
t_by_cat <- cog_spending(gov, 2019L, expenditure_concept = "total")
|
||||
prov_by_cat <- cog_explain(t_by_cat, format = "list")
|
||||
expect_false(is.na(prov_by_cat$expenditure_concept_direct_suppressed))
|
||||
})
|
||||
|
||||
test_that('"All Categories" still signposts coverage gaps (finding 6, final whole-branch review)', {
|
||||
# .build_suggestions()'s candidate sub-select used to be keyed on
|
||||
# `category`, e.g. `WHERE category IN ('All Categories')`. Since
|
||||
# .ALL_CATEGORIES is never itself a row in summary_categories.category,
|
||||
# that sub-select always came back empty in all-categories mode, so
|
||||
# `candidates` was empty and .build_suggestions() short-circuited to
|
||||
# list() -- coverage signposting was structurally impossible for the one
|
||||
# mode whose whole selling point is "you cannot sum the wrong scope"
|
||||
# (uscogdata#9's entire point, silently defeated).
|
||||
#
|
||||
# AL state government, FY2011, category = "Corrections": this category has
|
||||
# no legacy leaf rows in FY2011 (aggregate-flagged E04/E05 family), so the
|
||||
# per-category query returns 0 rows and 3 recipe-hint suggestions fire
|
||||
# (empty_year path). All-categories mode does not have an empty year --
|
||||
# the government has other primary spending in FY2011 -- but the same
|
||||
# suppressed Corrections dollars are still excluded from the summed total,
|
||||
# so the fix (scoping the candidate sub-select by subtype_col/subtype_scope
|
||||
# instead of by category, symmetric with .build_verb_sql()) must still
|
||||
# surface them via the suppressed_component path.
|
||||
gov <- "010000226085"
|
||||
|
||||
by_cat <- suppressMessages(cog_spending(gov, 2011L, category = "Corrections"))
|
||||
sugg_by_cat <- cog_explain(by_cat, format = "list")$suggestions
|
||||
expect_gt(length(sugg_by_cat), 0L)
|
||||
|
||||
all_cat <- suppressMessages(cog_spending(gov, 2011L, category = "All Categories"))
|
||||
sugg_all_cat <- cog_explain(all_cat, format = "list")$suggestions
|
||||
expect_gt(length(sugg_all_cat), 0L)
|
||||
|
||||
# The same Corrections recipe that fired per-category must also fire in
|
||||
# all-categories mode -- not just some unrelated recipe.
|
||||
ids_by_cat <- vapply(sugg_by_cat, function(s) s$recipe_id %||% "", character(1))
|
||||
ids_all_cat <- vapply(sugg_all_cat, function(s) s$recipe_id %||% "", character(1))
|
||||
expect_true("corrections_combined" %in% ids_by_cat)
|
||||
expect_true("corrections_combined" %in% ids_all_cat)
|
||||
|
||||
# In all-categories mode the government DOES have other primary spending
|
||||
# in FY2011 (the year itself is not a gap), so the suggestion can only have
|
||||
# fired via the suppressed_component path, not empty_year.
|
||||
corr_all <- sugg_all_cat[[which(ids_all_cat == "corrections_combined")]]
|
||||
expect_identical(corr_all$trigger, "suppressed_component")
|
||||
expect_gt(corr_all$suppressed_amount, 0)
|
||||
})
|
||||
|
||||
test_that('"All Categories" candidate scoping is symmetric with .build_verb_sql() -- subtype, not category', {
|
||||
# Direct assertion on the mechanism itself (finding 6): in all-categories
|
||||
# mode .build_suggestions() must scope its candidate recipe sub-select by
|
||||
# subtype_col/subtype_scope, not by the literal "All Categories" value.
|
||||
# Passing all_categories = FALSE with the identical category value proves
|
||||
# the branch -- not merely the subtype_col/subtype_scope arguments' mere
|
||||
# presence -- is what changes the query.
|
||||
con <- uscogdata:::.ensure_session()
|
||||
|
||||
none <- uscogdata:::.build_suggestions(
|
||||
con, cohort = uscogdata:::.make_cohort("010000226085"), years = 2011L,
|
||||
category = "All Categories", result = NULL, basis = "harmonized",
|
||||
flow_prefixes = c("E", "F", "G"),
|
||||
long_view = "spending_long_harmonized",
|
||||
all_categories = FALSE,
|
||||
subtype_col = "spend_subtype",
|
||||
subtype_scope = c("operations", "capital", "assistance")
|
||||
)
|
||||
expect_length(none, 0L)
|
||||
|
||||
scoped <- uscogdata:::.build_suggestions(
|
||||
con, cohort = uscogdata:::.make_cohort("010000226085"), years = 2011L,
|
||||
category = "All Categories", result = NULL, basis = "harmonized",
|
||||
flow_prefixes = c("E", "F", "G"),
|
||||
long_view = "spending_long_harmonized",
|
||||
all_categories = TRUE,
|
||||
subtype_col = "spend_subtype",
|
||||
subtype_scope = c("operations", "capital", "assistance")
|
||||
)
|
||||
expect_gt(length(scoped), 0L)
|
||||
})
|
||||
@@ -66,7 +66,7 @@ test_that("inst/sql/26-balance_long.sql enforces NOT is_aggregate (real SQL text
|
||||
sql_dir <- system.file("sql", package = "uscogdata")
|
||||
.read_view_sql <- function(filename) {
|
||||
txt <- paste(readLines(file.path(sql_dir, filename), warn = FALSE), collapse = "\n")
|
||||
uscogdata:::.render_view_sql(txt, paste0(tmp, "/"))
|
||||
gsub("\\{url\\}", paste0(tmp, "/"), txt, fixed = FALSE)
|
||||
}
|
||||
|
||||
con <- DBI::dbConnect(duckdb::duckdb())
|
||||
|
||||
@@ -29,9 +29,7 @@ test_that("cog_categories(type = 'spending') returns only expenditure rows", {
|
||||
# joined with the I/Q/Y flow batch -- the last two characters of Census's
|
||||
# expenditure taxonomy. `interest` is what makes the three-concept model
|
||||
# computable: primary = direct minus debt service.
|
||||
# Exclude pseudo-category which has NA for subtype
|
||||
r_crosswalk <- r[r$category != "All Categories", ]
|
||||
expect_true(all(r_crosswalk$subtype %in%
|
||||
expect_true(all(r$subtype %in%
|
||||
c("operations", "capital", "intergovernmental", "assistance",
|
||||
"interest", "insurance_benefits")))
|
||||
})
|
||||
@@ -56,9 +54,7 @@ test_that("cog_categories(type = 'revenue') returns only revenue rows", {
|
||||
# plus the employee-retirement X codes), utility (A91-A94) and liquor store
|
||||
# (A90) revenue by definition, which is what makes both of its published
|
||||
# revenue concepts computable -- see `revenue_concept` in `?cog_revenue`.
|
||||
# Exclude pseudo-category which has NA for subtype
|
||||
r_crosswalk <- r[r$category != "All Categories", ]
|
||||
expect_true(all(r_crosswalk$subtype %in%
|
||||
expect_true(all(r$subtype %in%
|
||||
c("own_source", "federal", "state", "local_aid",
|
||||
"insurance_trust", "utility", "liquor_store")))
|
||||
})
|
||||
@@ -73,8 +69,6 @@ test_that("cog_categories(pattern = ...) filters case-insensitively", {
|
||||
test_that("cog_categories has one row per (category, subtype)", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_categories()
|
||||
# Exclude pseudo-category which is not a crosswalk entry
|
||||
r <- r[r$category != "All Categories", ]
|
||||
key <- paste(r$category, r$subtype, sep = "|")
|
||||
expect_equal(length(key), length(unique(key)))
|
||||
})
|
||||
@@ -82,8 +76,6 @@ test_that("cog_categories has one row per (category, subtype)", {
|
||||
test_that("cog_categories item_codes is non-empty comma-separated string", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_categories()
|
||||
# Exclude pseudo-category which has NA for n_codes and item_codes
|
||||
r <- r[r$category != "All Categories", ]
|
||||
expect_true(all(nzchar(r$item_codes)))
|
||||
expect_true(all(r$n_codes >= 1L))
|
||||
# n_codes should equal count of commas + 1
|
||||
|
||||
@@ -1,218 +0,0 @@
|
||||
# The money verbs accepting a cohort by predicate (uscogdata#58).
|
||||
#
|
||||
# The load-bearing property is EQUIVALENCE: naming the same set of governments
|
||||
# by id and by state/type must return the same rows. Everything else here is
|
||||
# about the ways that equivalence could silently break -- the postal/FIPS
|
||||
# translation, the intersection rule, and provenance no longer having an id
|
||||
# list to describe.
|
||||
#
|
||||
# Fixture cohorts used: RI (fips 44) cities = 8 governments, DE (fips 10)
|
||||
# counties = 3. Small on purpose; the size of the win is measured against the
|
||||
# production corpus, not here.
|
||||
|
||||
# --- equivalence -----------------------------------------------------------
|
||||
|
||||
test_that("a predicate cohort returns exactly what the same ids return", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
ids <- cog_gov_search(NULL, state = "RI", type = "city")$canonical_govid
|
||||
expect_gt(length(ids), 1L)
|
||||
|
||||
by_id <- cog_spending(govid = ids, years = 2019)
|
||||
by_pred <- cog_spending(years = 2019, state = "RI", type = "city")
|
||||
|
||||
# Compare the data itself, ignoring the provenance attribute -- which is
|
||||
# SUPPOSED to differ (see the scope tests below).
|
||||
expect_equal(
|
||||
as.data.frame(by_id[order(by_id$canonical_govid, by_id$category), ]),
|
||||
as.data.frame(by_pred[order(by_pred$canonical_govid, by_pred$category), ]),
|
||||
ignore_attr = TRUE
|
||||
)
|
||||
expect_gt(nrow(by_pred), 0L)
|
||||
})
|
||||
})
|
||||
|
||||
test_that("equivalence holds for cog_revenue()", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
ids <- cog_gov_search(NULL, state = "DE", type = "county")$canonical_govid
|
||||
by_id <- cog_revenue(govid = ids, years = 2019)
|
||||
by_pred <- cog_revenue(years = 2019, state = "DE", type = "county")
|
||||
expect_equal(nrow(by_id), nrow(by_pred))
|
||||
expect_equal(sum(by_id$amt_nominal), sum(by_pred$amt_nominal))
|
||||
})
|
||||
})
|
||||
|
||||
test_that("equivalence holds for cog_balances()", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
ids <- cog_gov_search(NULL, state = "DE", type = "county")$canonical_govid
|
||||
by_id <- cog_balances(govid = ids, years = 2019)
|
||||
by_pred <- cog_balances(years = 2019, state = "DE", type = "county")
|
||||
expect_equal(nrow(by_id), nrow(by_pred))
|
||||
expect_equal(sum(by_id$amt_nominal), sum(by_pred$amt_nominal))
|
||||
})
|
||||
})
|
||||
|
||||
test_that("equivalence survives per_capita, adjust_to_year and pagination", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
ids <- cog_gov_search(NULL, state = "RI", type = "city")$canonical_govid
|
||||
|
||||
# per_capita now keys its population lookup on the rows in the result
|
||||
# rather than on the requested cohort; these must stay identical.
|
||||
by_id <- cog_spending(govid = ids, years = 2019, per_capita = TRUE,
|
||||
adjust_to_year = 2020)
|
||||
by_pred <- cog_spending(years = 2019, state = "RI", type = "city",
|
||||
per_capita = TRUE, adjust_to_year = 2020)
|
||||
expect_equal(by_id$amt_per_capita_nominal, by_pred$amt_per_capita_nominal)
|
||||
expect_equal(by_id$amt_per_capita_real, by_pred$amt_per_capita_real)
|
||||
expect_equal(by_id$pop_source, by_pred$pop_source)
|
||||
|
||||
paged_id <- cog_spending(govid = ids, years = 2019, limit = 5, offset = 5)
|
||||
paged_pred <- cog_spending(years = 2019, state = "RI", type = "city",
|
||||
limit = 5, offset = 5)
|
||||
expect_equal(as.data.frame(paged_id), as.data.frame(paged_pred),
|
||||
ignore_attr = TRUE)
|
||||
expect_identical(attr(paged_id, "total_rows"), attr(paged_pred, "total_rows"))
|
||||
})
|
||||
})
|
||||
|
||||
test_that("a state-only predicate spans every type in that state", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
ids <- cog_gov_search(NULL, state = "DE", type = NULL)$canonical_govid
|
||||
by_id <- cog_spending(govid = ids, years = 2019)
|
||||
by_pred <- cog_spending(years = 2019, state = "DE")
|
||||
expect_equal(nrow(by_id), nrow(by_pred))
|
||||
})
|
||||
})
|
||||
|
||||
# --- the postal/FIPS trap --------------------------------------------------
|
||||
|
||||
test_that("a postal abbreviation resolves to rows, not to silence", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
# canonical_fips_xwalk.fips_state holds "44", not "RI". A predicate built
|
||||
# from the raw parameter matches nothing and returns an empty result that
|
||||
# reads as "these governments reported nothing" -- the exact trap cog-api
|
||||
# hit. A zero-row result here is the regression.
|
||||
r <- cog_spending(years = 2019, state = "RI", type = "city")
|
||||
expect_gt(nrow(r), 0L)
|
||||
|
||||
# And the FIPS form is accepted as the same cohort.
|
||||
expect_equal(nrow(cog_spending(years = 2019, state = "44", type = "city")),
|
||||
nrow(r))
|
||||
})
|
||||
})
|
||||
|
||||
test_that("an unknown state abbreviation aborts with a message that names the problem", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
# Regression: `.state_abbrev_to_fips` is a named character vector, so `[[`
|
||||
# on an absent name threw base R's "subscript out of bounds" and the
|
||||
# curated message was unreachable.
|
||||
expect_error(cog_spending(years = 2019, state = "ZZ"),
|
||||
class = "uscogdata_unknown_state")
|
||||
expect_error(cog_spending(years = 2019, state = "ZZ"),
|
||||
"Unknown state abbreviation")
|
||||
})
|
||||
})
|
||||
|
||||
# --- naming the cohort -----------------------------------------------------
|
||||
|
||||
test_that("naming no cohort at all is refused", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
expect_error(cog_spending(years = 2019), class = "uscogdata_no_cohort")
|
||||
expect_error(cog_revenue(years = 2019), class = "uscogdata_no_cohort")
|
||||
expect_error(cog_balances(years = 2019), class = "uscogdata_no_cohort")
|
||||
})
|
||||
})
|
||||
|
||||
test_that("an empty govid vector still fails as it always did", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
# Supplied-but-empty is a caller error, not "cohort named some other way".
|
||||
expect_error(cog_spending(character(0), 2019),
|
||||
"must be a non-empty character vector")
|
||||
})
|
||||
})
|
||||
|
||||
test_that("govid and state/type together intersect", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
cities <- cog_gov_search(NULL, state = "RI", type = "city")$canonical_govid
|
||||
counties <- cog_gov_search(NULL, state = "DE", type = "county")$canonical_govid
|
||||
|
||||
# The documented rule: the governments in `govid` that ALSO match the
|
||||
# predicate -- never one silently taking precedence over the other.
|
||||
both <- cog_spending(govid = c(cities, counties), years = 2019,
|
||||
state = "RI", type = "city")
|
||||
only <- cog_spending(govid = cities, years = 2019)
|
||||
expect_equal(as.data.frame(both), as.data.frame(only), ignore_attr = TRUE)
|
||||
|
||||
# A disjoint intersection is empty, not "whichever one won".
|
||||
none <- cog_spending(govid = counties, years = 2019,
|
||||
state = "RI", type = "city")
|
||||
expect_equal(nrow(none), 0L)
|
||||
})
|
||||
})
|
||||
|
||||
# --- provenance ------------------------------------------------------------
|
||||
|
||||
test_that("a govid cohort's provenance scope is untouched", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
ids <- cog_gov_search(NULL, state = "DE", type = "county")$canonical_govid
|
||||
prov <- attr(cog_spending(govid = ids, years = 2019), "provenance")
|
||||
expect_setequal(prov$scope$govids_found, ids)
|
||||
expect_length(prov$scope$govids_missing, 0L)
|
||||
# No cohort block: govids_found already describes this cohort exactly.
|
||||
expect_null(prov$scope$cohort)
|
||||
})
|
||||
})
|
||||
|
||||
test_that("a predicate cohort describes itself instead of listing ids", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
prov <- attr(cog_spending(years = 2019, state = "DE", type = "county"),
|
||||
"provenance")
|
||||
# Deliberately NOT the resolved id list: a fleet-scale cohort would put
|
||||
# 20,000 govids into every response body.
|
||||
expect_length(prov$scope$govids_found, 0L)
|
||||
expect_length(prov$scope$govids_missing, 0L)
|
||||
expect_identical(prov$scope$cohort$state, "DE")
|
||||
expect_identical(prov$scope$cohort$type, "county")
|
||||
expect_identical(prov$scope$cohort$n_governments, 3L)
|
||||
})
|
||||
})
|
||||
|
||||
test_that("the cohort block counts the intersection, not the predicate alone", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
counties <- cog_gov_search(NULL, state = "DE", type = "county")$canonical_govid
|
||||
prov <- attr(
|
||||
cog_spending(govid = counties[1], years = 2019, state = "DE", type = "county"),
|
||||
"provenance"
|
||||
)
|
||||
expect_identical(prov$scope$cohort$n_governments, 1L)
|
||||
})
|
||||
})
|
||||
|
||||
# --- the SQL actually changed ----------------------------------------------
|
||||
|
||||
test_that("a predicate cohort never renders the ids into the query", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
ids <- cog_gov_search(NULL, state = "RI", type = "city")$canonical_govid
|
||||
prov <- attr(cog_spending(years = 2019, state = "RI", type = "city"),
|
||||
"provenance")
|
||||
sql <- prov$sql_query %||% prov$sql
|
||||
skip_if(is.null(sql), "provenance carries no SQL for this verb")
|
||||
# The whole point: cohort size does not enter the SQL string.
|
||||
for (id in ids) expect_false(grepl(id, sql, fixed = TRUE))
|
||||
expect_match(sql, "SELECT canonical_govid FROM canonical_fips_xwalk",
|
||||
fixed = TRUE)
|
||||
})
|
||||
})
|
||||
@@ -1,83 +0,0 @@
|
||||
# A cohort is how every verb names the set of governments it queries. It can be
|
||||
# named by explicit id, by a predicate over canonical_fips_xwalk, or by both
|
||||
# (intersection). These tests cover the SQL construction itself -- pure string
|
||||
# building, no corpus needed -- because that is where the postal/FIPS trap and
|
||||
# the 40k-literal blowup both live.
|
||||
|
||||
test_that(".make_cohort() keeps an explicit id vector as ids", {
|
||||
ch <- .make_cohort(govid = c("550000227544", "060000000001"))
|
||||
expect_identical(ch$ids, c("550000227544", "060000000001"))
|
||||
expect_null(ch$state_fips)
|
||||
expect_null(ch$type_int)
|
||||
expect_false(.cohort_by_predicate(ch))
|
||||
})
|
||||
|
||||
test_that(".make_cohort() translates a postal abbreviation to FIPS", {
|
||||
# The trap this whole issue exists to avoid: canonical_fips_xwalk.fips_state
|
||||
# holds "55", not "WI". A predicate written against the raw parameter matches
|
||||
# nothing and returns an empty result that reads as "reported nothing".
|
||||
ch <- .make_cohort(state = "WI")
|
||||
expect_identical(ch$state_fips, "55")
|
||||
expect_true(.cohort_by_predicate(ch))
|
||||
})
|
||||
|
||||
test_that(".make_cohort() translates a type label to its integer code", {
|
||||
expect_identical(.make_cohort(type = "city")$type_int, 2L)
|
||||
expect_identical(.make_cohort(type = "state")$type_int, 0L)
|
||||
expect_identical(.make_cohort(type = 1)$type_int, 1L)
|
||||
})
|
||||
|
||||
test_that(".make_cohort() reuses the search verb's coercers for invalid input", {
|
||||
expect_error(.make_cohort(state = "ZZ"), "Unknown state abbreviation")
|
||||
expect_error(.make_cohort(type = "special_district"), "Unknown type")
|
||||
# Out-of-scope types (4 = special district, 5 = school district) are refused
|
||||
# by the numeric branch, with the v0.1-scope message.
|
||||
expect_error(.make_cohort(type = 4), "type must be 0, 1, 2, or 3")
|
||||
})
|
||||
|
||||
test_that(".make_cohort() rejects naming no cohort at all", {
|
||||
expect_error(.make_cohort(), class = "uscogdata_no_cohort")
|
||||
})
|
||||
|
||||
test_that(".cohort_sql() renders an id cohort as a literal IN list", {
|
||||
sql <- .cohort_sql(.make_cohort(govid = c("a", "b")))
|
||||
expect_identical(sql, "canonical_govid IN ('a','b')")
|
||||
})
|
||||
|
||||
test_that(".cohort_sql() renders a predicate cohort as an xwalk subquery", {
|
||||
# The point of the issue: the cohort never becomes a literal list, so its
|
||||
# size does not enter the SQL string at all.
|
||||
sql <- .cohort_sql(.make_cohort(state = "WI", type = "city"))
|
||||
expect_match(sql, "SELECT canonical_govid FROM canonical_fips_xwalk", fixed = TRUE)
|
||||
expect_match(sql, "fips_state = '55'", fixed = TRUE)
|
||||
expect_match(sql, "govs_type = 2", fixed = TRUE)
|
||||
expect_false(grepl("'WI'", sql, fixed = TRUE))
|
||||
})
|
||||
|
||||
test_that(".cohort_sql() renders ids and a predicate as an intersection", {
|
||||
sql <- .cohort_sql(.make_cohort(govid = c("a", "b"), type = "city"))
|
||||
expect_match(sql, "canonical_govid IN ('a','b')", fixed = TRUE)
|
||||
expect_match(sql, "AND canonical_govid IN (SELECT", fixed = TRUE)
|
||||
})
|
||||
|
||||
test_that(".cohort_sql() honours a column alias", {
|
||||
# Several call sites join the xwalk under an alias (`l.`, `x.`, `v.`), so the
|
||||
# predicate has to be able to name the qualified column.
|
||||
sql <- .cohort_sql(.make_cohort(state = "WI"), col = "l.canonical_govid")
|
||||
expect_match(sql, "l.canonical_govid IN (SELECT", fixed = TRUE)
|
||||
# The subquery's own column stays unqualified -- it selects from the xwalk,
|
||||
# not from the outer relation.
|
||||
expect_match(sql, "SELECT canonical_govid FROM", fixed = TRUE)
|
||||
})
|
||||
|
||||
test_that(".cohort_sql() escapes quotes in ids", {
|
||||
sql <- .cohort_sql(.make_cohort(govid = "o'brien"))
|
||||
expect_match(sql, "'o''brien'", fixed = TRUE)
|
||||
})
|
||||
|
||||
test_that("a predicate cohort's SQL does not grow with cohort size", {
|
||||
# The regression this guards: 20,106 ids rendered to a 301,591-character
|
||||
# IN list, embedded in 5-8 statements per call.
|
||||
wide <- .cohort_sql(.make_cohort(state = "CA", type = "city"))
|
||||
expect_lt(nchar(wide), 200L)
|
||||
})
|
||||
@@ -64,109 +64,3 @@ test_that(".resolve_url does not invent a slash for an empty setting", {
|
||||
withr::local_options(uscogdata.url = "")
|
||||
expect_equal(.resolve_url(), "")
|
||||
})
|
||||
|
||||
test_that("the default corpus URL is real, not a placeholder", {
|
||||
# setup.R points USCOGDATA_URL at the bundled fixture for the whole suite,
|
||||
# so both the env var and the option have to be cleared to see the default.
|
||||
withr::local_envvar(USCOGDATA_URL = NA)
|
||||
withr::local_options(uscogdata.url = NULL)
|
||||
url <- .resolve_url()
|
||||
expect_false(grepl("REPLACE_WITH", url, fixed = TRUE))
|
||||
expect_match(url, "^https://")
|
||||
expect_match(url, "/$")
|
||||
})
|
||||
|
||||
test_that("an explicitly-set sentinel URL still aborts", {
|
||||
# The guard must survive the default change: a user who half-edited a
|
||||
# copied config still gets the actionable error.
|
||||
withr::local_envvar(
|
||||
USCOGDATA_URL = "https://other.example/s/REPLACE_WITH_SHARE_TOKEN/x/"
|
||||
)
|
||||
expect_error(
|
||||
.check_url_configured(.resolve_url()),
|
||||
class = "uscogdata_url_not_configured"
|
||||
)
|
||||
})
|
||||
|
||||
test_that("DESCRIPTION carries release metadata", {
|
||||
skip_if_no_source_tree("DESCRIPTION")
|
||||
d <- read.dcf(source_tree_path("DESCRIPTION"))
|
||||
fields <- colnames(d)
|
||||
|
||||
expect_true(all(c("URL", "BugReports") %in% fields))
|
||||
expect_match(d[1, "Authors@R"], "Knowles", fixed = TRUE)
|
||||
expect_match(d[1, "Authors@R"], "0000-0003-0005-9478", fixed = TRUE)
|
||||
expect_match(d[1, "Authors@R"], "Civilytics Consulting LLC", fixed = TRUE)
|
||||
|
||||
# The gate in .validate_schema() accepts up to 7 and the published corpus
|
||||
# IS 7; DESCRIPTION must not claim otherwise.
|
||||
expect_equal(as.integer(d[1, "MaxCorpusSchema"]), 7L)
|
||||
|
||||
# Authors@R must actually parse -- a malformed person() call is only
|
||||
# caught at citation()/build time otherwise.
|
||||
people <- eval(parse(text = d[1, "Authors@R"]))
|
||||
expect_s3_class(people, "person")
|
||||
expect_true("cre" %in% unlist(lapply(people, function(p) p$role)))
|
||||
})
|
||||
|
||||
test_that("LICENSE and LICENSE.md name the same copyright holder", {
|
||||
skip_if_no_source_tree("LICENSE", "LICENSE.md")
|
||||
holder <- sub("^COPYRIGHT HOLDER:\\s*", "",
|
||||
grep("^COPYRIGHT HOLDER:", readLines(source_tree_path("LICENSE"),
|
||||
warn = FALSE), value = TRUE))
|
||||
full <- paste(readLines(source_tree_path("LICENSE.md"), warn = FALSE), collapse = "\n")
|
||||
|
||||
expect_equal(holder, "Civilytics Consulting LLC")
|
||||
expect_match(full, holder, fixed = TRUE)
|
||||
# usethis::use_mit_license() writes LICENSE.md but leaves an existing
|
||||
# LICENSE alone, which is how the two came to disagree in the first place.
|
||||
expect_match(full, "MIT License", fixed = TRUE)
|
||||
})
|
||||
|
||||
test_that("vignettes are not excluded from the build", {
|
||||
skip_if_no_source_tree(".Rbuildignore")
|
||||
ignore <- readLines(source_tree_path(".Rbuildignore"), warn = FALSE)
|
||||
expect_false(any(grepl("^\\^vignettes\\$$", ignore)))
|
||||
# The fixture is what lets R CMD check run offline with no credentials on
|
||||
# r-universe and GitHub Actions. It must never be excluded.
|
||||
expect_false(any(grepl("fixture_corpus", ignore, fixed = TRUE)))
|
||||
# doc/ and Meta/ ARE build artefacts of devtools::build_vignettes() and must
|
||||
# stay excluded -- R CMD build regenerates inst/doc/ from vignettes/ on its
|
||||
# own, and leaving them in earns a "non-standard file at top level" NOTE.
|
||||
expect_true(any(grepl("^\\^doc\\$$", ignore)))
|
||||
expect_true(any(grepl("^\\^Meta\\$$", ignore)))
|
||||
})
|
||||
|
||||
test_that("_pkgdown.yml indexes every exported topic", {
|
||||
skip_if_no_source_tree("_pkgdown.yml", "NAMESPACE")
|
||||
exports <- grep("^export\\(", readLines(source_tree_path("NAMESPACE"), warn = FALSE),
|
||||
value = TRUE)
|
||||
exports <- sub("^export\\((.*)\\)$", "\\1", exports)
|
||||
yml <- paste(readLines(source_tree_path("_pkgdown.yml"), warn = FALSE), collapse = "\n")
|
||||
missing <- exports[!vapply(exports,
|
||||
function(e) grepl(paste0("\\b", e, "\\b"), yml),
|
||||
logical(1))]
|
||||
# pkgdown errors on topics missing from the index, so an unlisted export
|
||||
# means the docs site does not build at all.
|
||||
expect_equal(missing, character(0))
|
||||
})
|
||||
|
||||
test_that("README is written for a stranger, not a repo insider", {
|
||||
skip_if_no_source_tree("README.md")
|
||||
r <- paste(readLines(source_tree_path("README.md"), warn = FALSE), collapse = "\n")
|
||||
|
||||
# No paths that only resolve inside a maintainer's checkout.
|
||||
expect_false(grepl("../cog_pipeline", r, fixed = TRUE))
|
||||
# A real, uncommented install line.
|
||||
expect_match(r, "install.packages", fixed = TRUE)
|
||||
expect_false(grepl("# pak::pkg_install", r, fixed = TRUE))
|
||||
# The errata most likely to produce a plausible-looking wrong answer.
|
||||
expect_match(r, "full US dollars", fixed = TRUE)
|
||||
# The release advice that conflicts with public CI is gone.
|
||||
expect_false(grepl("Rbuildignore", r, fixed = TRUE))
|
||||
# Both read paths documented.
|
||||
expect_match(r, "cog_mirror", fixed = TRUE)
|
||||
# cog_spending() has no default for `years`; a quickstart that omits it
|
||||
# errors on the reader's first call.
|
||||
expect_match(r, "years\\s*=", perl = TRUE)
|
||||
})
|
||||
|
||||
@@ -1,134 +0,0 @@
|
||||
# tests/testthat/test-duckdb-limits.R
|
||||
#
|
||||
# uscogdata#60. cog_open() used to connect with a bare dbConnect() and set no
|
||||
# resource pragmas, so DuckDB claimed every visible core. That is right for one
|
||||
# interactive session on a dedicated machine and wrong for a server: cog-api
|
||||
# runs two replicas on an 8-core host budgeted 4, and without a cap each
|
||||
# replica independently claims all 8 and they fight.
|
||||
#
|
||||
# The consumer-side workaround this replaces reached into the namespace at
|
||||
# boot -- getFromNamespace(".ensure_session", "uscogdata")() followed by a
|
||||
# manual SET threads -- which depends on a private name AND on the session
|
||||
# already being open.
|
||||
#
|
||||
# The load-bearing property is the NEGATIVE one: unset must emit no pragma at
|
||||
# all, so an unconfigured session is byte-identical to pre-#60 behaviour.
|
||||
|
||||
# Open a session under a given configuration and read a DuckDB setting back.
|
||||
# Each call closes first, because both settings are session-scoped: an
|
||||
# already-open connection would be reused by .ensure_session() and report the
|
||||
# PREVIOUS test's value, which is exactly the false pass to avoid here.
|
||||
setting_under <- function(setting, envvars = character(0), opts = list()) {
|
||||
uscogdata:::cog_close()
|
||||
on.exit(uscogdata:::cog_close(), add = TRUE)
|
||||
withr::with_envvar(envvars, {
|
||||
withr::with_options(opts, {
|
||||
con <- uscogdata:::cog_open()
|
||||
DBI::dbGetQuery(
|
||||
con, sprintf("SELECT current_setting('%s') AS v", setting)
|
||||
)$v[[1]]
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
test_that("USCOGDATA_DUCKDB_THREADS caps the connection's thread count", {
|
||||
skip_if_no_corpus()
|
||||
expect_equal(
|
||||
as.integer(setting_under("threads", c(USCOGDATA_DUCKDB_THREADS = "2"))),
|
||||
2L
|
||||
)
|
||||
})
|
||||
|
||||
test_that("the option spelling works, and the env var beats it", {
|
||||
skip_if_no_corpus()
|
||||
expect_equal(
|
||||
as.integer(setting_under("threads",
|
||||
c(USCOGDATA_DUCKDB_THREADS = NA),
|
||||
list(uscogdata.duckdb_threads = 3L))),
|
||||
3L
|
||||
)
|
||||
# Same precedence .cfg() gives every other setting: env var > option.
|
||||
expect_equal(
|
||||
as.integer(setting_under("threads",
|
||||
c(USCOGDATA_DUCKDB_THREADS = "1"),
|
||||
list(uscogdata.duckdb_threads = 3L))),
|
||||
1L
|
||||
)
|
||||
})
|
||||
|
||||
test_that("unset leaves DuckDB's own default in place", {
|
||||
skip_if_no_corpus()
|
||||
# Not asserting a specific number -- the default is core-count-dependent and
|
||||
# a literal would fail on a different machine. The claim is that NO pragma
|
||||
# was issued, so the session sees whatever DuckDB would have chosen on its
|
||||
# own. Compared against a plain connection opened the pre-#60 way.
|
||||
unset <- setting_under("threads",
|
||||
c(USCOGDATA_DUCKDB_THREADS = NA),
|
||||
list(uscogdata.duckdb_threads = NULL))
|
||||
bare <- local({
|
||||
con <- DBI::dbConnect(duckdb::duckdb())
|
||||
on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE)
|
||||
DBI::dbGetQuery(con, "SELECT current_setting('threads') AS v")$v[[1]]
|
||||
})
|
||||
expect_equal(as.integer(unset), as.integer(bare))
|
||||
})
|
||||
|
||||
test_that("USCOGDATA_DUCKDB_MEMORY_LIMIT is applied", {
|
||||
skip_if_no_corpus()
|
||||
v <- setting_under("memory_limit", c(USCOGDATA_DUCKDB_MEMORY_LIMIT = "2GB"))
|
||||
# DuckDB does not echo back the string it was given: it stores bytes and
|
||||
# reports BINARY units, so "2GB" (2e9 bytes) comes back as "1.8 GiB". Assert
|
||||
# the magnitude it actually means rather than the spelling this package sent
|
||||
# -- matching on "2" passes for the wrong reason and fails on the right one.
|
||||
expect_match(as.character(v), "GiB", fixed = TRUE)
|
||||
# DuckDB also truncates the display to one decimal ("1.8 GiB" for 1.863), so
|
||||
# the tolerance covers rounding, not slack in the setting itself.
|
||||
gib <- as.numeric(sub("\\s*GiB$", "", as.character(v)))
|
||||
expect_equal(gib, 2e9 / 1024^3, tolerance = 0.05)
|
||||
})
|
||||
|
||||
# --- Validation -------------------------------------------------------------
|
||||
# .cfg() returns an env var as CHARACTER. Without coercion here,
|
||||
# sprintf("SET threads TO %d", "4") aborts inside the connection path with an
|
||||
# error about the pragma rather than about the setting the operator got wrong.
|
||||
|
||||
test_that(".resolve_duckdb_threads coerces a character env var to integer", {
|
||||
withr::local_envvar(USCOGDATA_DUCKDB_THREADS = "4")
|
||||
expect_identical(uscogdata:::.resolve_duckdb_threads(), 4L)
|
||||
})
|
||||
|
||||
test_that(".resolve_duckdb_threads returns NULL when unset or empty", {
|
||||
withr::local_options(uscogdata.duckdb_threads = NULL)
|
||||
withr::local_envvar(USCOGDATA_DUCKDB_THREADS = NA)
|
||||
expect_null(uscogdata:::.resolve_duckdb_threads())
|
||||
|
||||
withr::local_envvar(USCOGDATA_DUCKDB_THREADS = "")
|
||||
expect_null(uscogdata:::.resolve_duckdb_threads())
|
||||
})
|
||||
|
||||
test_that(".resolve_duckdb_threads rejects values that are not positive integers", {
|
||||
for (bad in c("0", "-1", "two", "1.5.2")) {
|
||||
withr::local_envvar(USCOGDATA_DUCKDB_THREADS = bad)
|
||||
expect_error(uscogdata:::.resolve_duckdb_threads(),
|
||||
class = "uscogdata_invalid_duckdb_threads")
|
||||
}
|
||||
})
|
||||
|
||||
test_that(".resolve_duckdb_memory_limit accepts size strings and rejects junk", {
|
||||
withr::local_envvar(USCOGDATA_DUCKDB_MEMORY_LIMIT = "4GB")
|
||||
expect_identical(uscogdata:::.resolve_duckdb_memory_limit(), "4GB")
|
||||
|
||||
withr::local_envvar(USCOGDATA_DUCKDB_MEMORY_LIMIT = "1.5GB")
|
||||
expect_identical(uscogdata:::.resolve_duckdb_memory_limit(), "1.5GB")
|
||||
|
||||
# A SQL fragment must not reach the connection as one.
|
||||
withr::local_envvar(USCOGDATA_DUCKDB_MEMORY_LIMIT = "4GB'; DROP TABLE x; --")
|
||||
expect_error(uscogdata:::.resolve_duckdb_memory_limit(),
|
||||
class = "uscogdata_invalid_duckdb_memory_limit")
|
||||
})
|
||||
|
||||
test_that(".apply_duckdb_limits issues no statement when both are NULL", {
|
||||
# The negative property, asserted directly rather than inferred: a connection
|
||||
# that would ERROR on any statement proves none was sent.
|
||||
expect_silent(uscogdata:::.apply_duckdb_limits(NULL, NULL, NULL))
|
||||
})
|
||||
@@ -315,17 +315,15 @@ test_that("a mis-scoped cog_spending() call never attaches an M/L counterpart to
|
||||
# (SB194, cog_pipeline#64), so the recipe stopped being a candidate there.
|
||||
# FL state carries a real FY2011 B47 amount, so this exercises the guard
|
||||
# against a suggestion that genuinely fires.
|
||||
#
|
||||
# Issue #34: "IG Federal" maps to B-prefixed codes in summary_categories
|
||||
# with category_type = 'revenue'. A spending verb (flow_prefixes E/F/G)
|
||||
# now scopes its candidate query by category_type = 'expenditure', so it
|
||||
# correctly finds NO candidates for this revenue-only category -- the
|
||||
# suggestion machinery cannot fire, and no M/L counterpart is attached.
|
||||
r <- suppressMessages(
|
||||
cog_spending("120000226351", years = c(2005, 2011), category = "IG Federal")
|
||||
)
|
||||
sugg <- attr(r, "provenance")$suggestions
|
||||
expect_length(sugg, 0L)
|
||||
expect_gt(length(sugg), 0L)
|
||||
ids <- vapply(sugg, function(s) s$recipe_id %||% "", character(1))
|
||||
expect_true("ig_federal_b47_wide" %in% ids)
|
||||
ig <- unlist(lapply(sugg, function(s) s$ig_recipe_id))
|
||||
expect_length(ig, 0L)
|
||||
})
|
||||
|
||||
test_that("C1: 'total' on a legacy aggregate-only family reports the IG-only figure honestly, not as Direct + IG", {
|
||||
|
||||
@@ -1,62 +0,0 @@
|
||||
# Network-gated. Set USCOGDATA_LIVE_TEST=true to run.
|
||||
#
|
||||
# This file exists because the defect fixed for 0.3.0 -- no remote corpus was
|
||||
# readable at all, because DuckDB cannot expand a glob over generic HTTP --
|
||||
# survived precisely because every other test path used a LOCAL corpus (the
|
||||
# bundled fixture), and so did the API in production (a host mount). Nothing
|
||||
# ever exercised the package the way a new user does.
|
||||
skip_live <- function() {
|
||||
testthat::skip_if_not(
|
||||
identical(tolower(Sys.getenv("USCOGDATA_LIVE_TEST", "")), "true"),
|
||||
"live-corpus test: set USCOGDATA_LIVE_TEST=true to run"
|
||||
)
|
||||
}
|
||||
|
||||
# The suite's setup.R pins USCOGDATA_URL to the bundled fixture, so reaching
|
||||
# the default requires clearing both the env var and the option.
|
||||
with_default_corpus <- function(code) {
|
||||
withr::local_envvar(
|
||||
USCOGDATA_URL = NA, USCOGDATA_FIXTURE_URL = NA,
|
||||
.local_envir = parent.frame()
|
||||
)
|
||||
withr::local_options(uscogdata.url = NULL, .local_envir = parent.frame())
|
||||
cog_close()
|
||||
withr::defer(cog_close(), envir = parent.frame())
|
||||
force(code)
|
||||
}
|
||||
|
||||
test_that("the package reads the public corpus with no configuration at all", {
|
||||
skip_live()
|
||||
with_default_corpus({
|
||||
g <- cog_gov_search(name = "Madison", state = "WI", type = 2)
|
||||
expect_gt(nrow(g), 0)
|
||||
|
||||
s <- cog_spending(g$canonical_govid[1], years = 2022)
|
||||
expect_gt(nrow(s), 0)
|
||||
expect_true(all(c("amt_nominal", "year", "category") %in% names(s)))
|
||||
|
||||
# Amounts are full dollars, already x1000. A city's annual spending is
|
||||
# millions, not thousands -- this catches a regression that dropped or
|
||||
# doubled the conversion.
|
||||
expect_gt(sum(s$amt_nominal, na.rm = TRUE), 1e6)
|
||||
|
||||
p <- attr(s, "provenance")
|
||||
expect_true(isTRUE(p$transformations$units_conversion$applied))
|
||||
expect_equal(p$transformations$units_conversion$multiplier, 1000)
|
||||
})
|
||||
})
|
||||
|
||||
test_that("a multi-decade query reads across many partitions", {
|
||||
skip_live()
|
||||
with_default_corpus({
|
||||
g <- cog_gov_search(name = "Madison", state = "WI", type = 2)
|
||||
# `years` is required on cog_spending() -- there is no full-history
|
||||
# default at the reader level (the API's /profile route supplies one).
|
||||
s <- cog_spending(g$canonical_govid[1], years = 2000:2022)
|
||||
# Enumeration builds one read_parquet() path per requested partition. If
|
||||
# the list were truncated, or silently collapsed to a single file, the
|
||||
# returned span is what catches it.
|
||||
expect_gt(diff(range(s$year)), 10)
|
||||
expect_gt(length(unique(s$year)), 5)
|
||||
})
|
||||
})
|
||||
@@ -1,97 +0,0 @@
|
||||
test_that(".long_files_sql enumerates every partition the manifest lists", {
|
||||
manifest <- list(files = list(long_partitions = list(
|
||||
list(year = 2011L, path = "data/long/year=2011/part-0.parquet"),
|
||||
list(year = 2012L, path = "data/long/year=2012/part-0.parquet")
|
||||
)))
|
||||
expect_equal(
|
||||
uscogdata:::.long_files_sql("https://example.org/corpus/", manifest),
|
||||
paste0(
|
||||
"['https://example.org/corpus/data/long/year=2011/part-0.parquet',",
|
||||
"'https://example.org/corpus/data/long/year=2012/part-0.parquet']"
|
||||
)
|
||||
)
|
||||
})
|
||||
|
||||
test_that(".long_files_sql falls back to the glob when no partition list is present", {
|
||||
# test-views.R registers views with a hand-built manifest that has no
|
||||
# `files` element. That must keep working: the glob is valid for the
|
||||
# local paths such a manifest is used with.
|
||||
expect_equal(
|
||||
uscogdata:::.long_files_sql("/tmp/corpus/", list(schema_version = 4L)),
|
||||
"'/tmp/corpus/data/long/**/*.parquet'"
|
||||
)
|
||||
expect_equal(
|
||||
uscogdata:::.long_files_sql("/tmp/corpus/", list(files = list(long_partitions = list()))),
|
||||
"'/tmp/corpus/data/long/**/*.parquet'"
|
||||
)
|
||||
})
|
||||
|
||||
test_that("the enumerated list matches the bundled fixture's partition count", {
|
||||
skip_if_no_corpus()
|
||||
m <- jsonlite::fromJSON(
|
||||
file.path(fixture_corpus_path(), "manifest.json"), simplifyVector = FALSE
|
||||
)
|
||||
out <- uscogdata:::.long_files_sql(fixture_corpus_path(), m)
|
||||
expect_equal(
|
||||
lengths(regmatches(out, gregexpr("part-0\\.parquet", out))),
|
||||
length(m$files$long_partitions)
|
||||
)
|
||||
})
|
||||
|
||||
test_that("no view SQL survives rendering with an unsubstituted token", {
|
||||
# Introducing {long_files} broke four test sites that had hand-rolled the
|
||||
# {url} substitution -- each failed with a DuckDB parser error on the
|
||||
# surviving brace. This asserts the whole SQL directory renders clean, so
|
||||
# a future token cannot reintroduce that silently.
|
||||
sql_dir <- system.file("sql", package = "uscogdata")
|
||||
for (f in list.files(sql_dir, pattern = "\\.sql$", full.names = TRUE)) {
|
||||
rendered <- uscogdata:::.render_view_sql(
|
||||
paste(readLines(f, warn = FALSE), collapse = "\n"), "/tmp/corpus/"
|
||||
)
|
||||
expect_false(grepl("\\{[a-z_]+\\}", rendered), label = basename(f))
|
||||
}
|
||||
})
|
||||
|
||||
test_that("registered `long` view reads through the enumerated list", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
con <- uscogdata:::.ensure_session()
|
||||
n <- DBI::dbGetQuery(con, "SELECT count(*) AS n FROM long")$n
|
||||
expect_gt(n, 0)
|
||||
yrs <- DBI::dbGetQuery(con, "SELECT DISTINCT year FROM long ORDER BY year")$year
|
||||
expect_true(all(c(2011, 2012, 2019, 2020) %in% yrs))
|
||||
})
|
||||
})
|
||||
|
||||
test_that("a Windows-style corpus path survives token substitution", {
|
||||
# gsub() in regex mode treats backslashes in the REPLACEMENT as escape
|
||||
# sequences and silently drops them, so a Windows path went in as
|
||||
# C:\Users\RUNNER\... and came out as C:UsersRUNNER..., after which every
|
||||
# DuckDB read failed with "No files found that match the pattern".
|
||||
#
|
||||
# That made a LOCAL corpus unreadable on Windows -- the bundled fixture
|
||||
# included, so the whole suite failed there -- while remote https URLs
|
||||
# worked fine, having no backslashes. It went unnoticed for the life of the
|
||||
# package because nothing ever ran on Windows.
|
||||
#
|
||||
# Reproducible on any platform: this is string handling, not a filesystem
|
||||
# behaviour, so it does not need a Windows runner to catch.
|
||||
win <- "C:\\Users\\RUNNER~1\\AppData\\Local\\Temp\\Rtmp123/"
|
||||
|
||||
out <- uscogdata:::.render_view_sql(
|
||||
"FROM read_parquet('{url}data/summary_categories.parquet')", win
|
||||
)
|
||||
expect_true(grepl("C:\\Users\\RUNNER~1\\AppData", out, fixed = TRUE))
|
||||
expect_false(grepl("C:Users", out, fixed = TRUE))
|
||||
|
||||
# The same must hold through the {long_files} path, which embeds the url
|
||||
# once per enumerated partition.
|
||||
manifest <- list(files = list(long_partitions = list(
|
||||
list(year = 2011L, path = "data/long/year=2011/part-0.parquet")
|
||||
)))
|
||||
out2 <- uscogdata:::.render_view_sql(
|
||||
"FROM read_parquet({long_files}, hive_partitioning = true)", win, manifest
|
||||
)
|
||||
expect_true(grepl("C:\\Users\\RUNNER~1\\AppData", out2, fixed = TRUE))
|
||||
expect_false(grepl("C:Users", out2, fixed = TRUE))
|
||||
})
|
||||
@@ -4,14 +4,12 @@
|
||||
# protect users from silent failures when USCOGDATA_URL is misconfigured
|
||||
# or returns non-JSON content.
|
||||
|
||||
test_that("cog_open aborts with actionable error when URL contains the sentinel", {
|
||||
test_that("cog_open aborts with actionable error when URL is the placeholder default", {
|
||||
uscogdata:::cog_close()
|
||||
on.exit(uscogdata:::cog_close(), add = TRUE)
|
||||
|
||||
# No longer the package default (that is the public HF corpus). This is a
|
||||
# user who copied a config template and did not finish editing it.
|
||||
sentinel_url <- "https://cloud.civilytics.org/s/REPLACE_WITH_SHARE_TOKEN/download/"
|
||||
withr::with_envvar(c(USCOGDATA_URL = sentinel_url), {
|
||||
placeholder <- "https://cloud.civilytics.org/s/REPLACE_WITH_SHARE_TOKEN/download/"
|
||||
withr::with_envvar(c(USCOGDATA_URL = placeholder), {
|
||||
expect_error(
|
||||
uscogdata:::cog_open(),
|
||||
class = "uscogdata_url_not_configured"
|
||||
@@ -37,10 +35,8 @@ test_that("placeholder guard error names both env var and option as remediation"
|
||||
uscogdata:::cog_close()
|
||||
on.exit(uscogdata:::cog_close(), add = TRUE)
|
||||
|
||||
# No longer the package default (that is the public HF corpus). This is a
|
||||
# user who copied a config template and did not finish editing it.
|
||||
sentinel_url <- "https://cloud.civilytics.org/s/REPLACE_WITH_SHARE_TOKEN/download/"
|
||||
withr::with_envvar(c(USCOGDATA_URL = sentinel_url), {
|
||||
placeholder <- "https://cloud.civilytics.org/s/REPLACE_WITH_SHARE_TOKEN/download/"
|
||||
withr::with_envvar(c(USCOGDATA_URL = placeholder), {
|
||||
msg <- tryCatch(uscogdata:::cog_open(), error = conditionMessage)
|
||||
expect_match(msg, "USCOGDATA_URL", fixed = TRUE)
|
||||
expect_match(msg, "uscogdata.url", fixed = TRUE)
|
||||
|
||||
@@ -214,246 +214,3 @@ test_that("no signposting under basis = 'raw'", {
|
||||
prov <- attr(r, "provenance")
|
||||
expect_length(prov$suggestions, 0L)
|
||||
})
|
||||
|
||||
# --- uscogdata#9: partial-coverage signposting ------------------------------
|
||||
|
||||
test_that("no recipe component is ever renamed by harmonization", {
|
||||
# The suppression trigger anti-joins the verb's long view on item_code.
|
||||
# That is only sound because harmonization never rewrites a recipe
|
||||
# component's code -- every component whose harmonized_code differs has
|
||||
# harmonized_code IS NULL (and is aggregate-flagged). If this ever fails,
|
||||
# .suppressed_components() would report reachable dollars as suppressed.
|
||||
skip_if_no_corpus()
|
||||
con <- uscogdata:::.ensure_session()
|
||||
n <- DBI::dbGetQuery(con,
|
||||
"SELECT COUNT(*) AS renamed FROM long
|
||||
WHERE item_code IN (SELECT DISTINCT component_code FROM harmonization_recipes)
|
||||
AND harmonized_code IS NOT NULL
|
||||
AND harmonized_code <> item_code")$renamed
|
||||
expect_equal(as.integer(n), 0L)
|
||||
})
|
||||
|
||||
test_that(".select_long_view maps annotated view bases to their long views", {
|
||||
expect_equal(
|
||||
uscogdata:::.select_long_view("spending_annotated", "harmonized"),
|
||||
"spending_long_harmonized")
|
||||
expect_equal(
|
||||
uscogdata:::.select_long_view("revenue_annotated", "harmonized"),
|
||||
"revenue_long_harmonized")
|
||||
expect_equal(
|
||||
uscogdata:::.select_long_view("spending_annotated", "raw"),
|
||||
"spending_long")
|
||||
})
|
||||
|
||||
test_that(".suppressed_components measures the E67/E68 dollars Public Welfare drops", {
|
||||
skip_if_no_corpus()
|
||||
con <- uscogdata:::.ensure_session()
|
||||
s <- uscogdata:::.suppressed_components(
|
||||
con,
|
||||
candidates = c("welfare_cash_e67_wide", "welfare_cash_e68_wide"),
|
||||
cohort = uscogdata:::.make_cohort("061037123085"), years = 2011L,
|
||||
long_view = "spending_long_harmonized",
|
||||
flow_prefixes = c("E", "F", "G"))
|
||||
|
||||
expect_s3_class(s, "tbl_df")
|
||||
expect_equal(nrow(s), 2L)
|
||||
s <- s[order(s$recipe_id), ]
|
||||
expect_equal(s$recipe_id, c("welfare_cash_e67_wide", "welfare_cash_e68_wide"))
|
||||
expect_equal(s$suppressed_amount, c(1803872000, 271589000))
|
||||
expect_equal(s$suppressed_codes, c("E67", "E68"))
|
||||
})
|
||||
|
||||
test_that(".suppressed_components finds nothing in a modern year", {
|
||||
skip_if_no_corpus()
|
||||
con <- uscogdata:::.ensure_session()
|
||||
s <- uscogdata:::.suppressed_components(
|
||||
con,
|
||||
candidates = c("welfare_cash_e67_wide", "welfare_cash_e68_wide"),
|
||||
cohort = uscogdata:::.make_cohort("061037123085"), years = 2019L,
|
||||
long_view = "spending_long_harmonized",
|
||||
flow_prefixes = c("E", "F", "G"))
|
||||
expect_equal(nrow(s), 0L)
|
||||
})
|
||||
|
||||
test_that(".suppressed_components rejects a long_view outside the allowlist", {
|
||||
skip_if_no_corpus()
|
||||
con <- uscogdata:::.ensure_session()
|
||||
expect_error(
|
||||
uscogdata:::.suppressed_components(
|
||||
con, candidates = "welfare_cash_e67_wide", cohort = uscogdata:::.make_cohort("061037123085"),
|
||||
years = 2011L, long_view = "long; DROP TABLE x",
|
||||
flow_prefixes = c("E", "F", "G")),
|
||||
class = "uscogdata_internal_error")
|
||||
})
|
||||
|
||||
test_that(".suppressed_components never measures a component from the other flow family (I1)", {
|
||||
# uscogdata#9 review, finding I1: without the flow_prefixes filter, a
|
||||
# candidate recipe entirely outside the calling verb's own flow family is
|
||||
# ALWAYS absent from that verb's view (by construction), so it was always
|
||||
# reported as "suppressed" -- fabricating a dollar claim. E67/E68 are
|
||||
# Public Welfare EXPENDITURE codes; scoping the measurement to revenue's
|
||||
# own flow_prefixes must find nothing for them.
|
||||
skip_if_no_corpus()
|
||||
con <- uscogdata:::.ensure_session()
|
||||
s <- uscogdata:::.suppressed_components(
|
||||
con,
|
||||
candidates = c("welfare_cash_e67_wide", "welfare_cash_e68_wide"),
|
||||
cohort = uscogdata:::.make_cohort("061037123085"), years = 2011L,
|
||||
long_view = "revenue_long_harmonized",
|
||||
flow_prefixes = c("T", "A", "U", "B", "C", "D"))
|
||||
expect_equal(nrow(s), 0L)
|
||||
})
|
||||
|
||||
test_that("uscogdata#9: Public Welfare signposts its suppressed E67/E68 dollars", {
|
||||
# The bug: E74/E79 return rows for FY2011, so there is no row-absence gap,
|
||||
# so nothing fired -- while E67 ($1,803,872,000) and E68 ($271,589,000) were
|
||||
# dropped for being aggregate-published. LA County reports $3,185,943,000
|
||||
# and omits $2,075,461,000, a 39% understatement, silently.
|
||||
skip_if_no_corpus()
|
||||
r <- suppressMessages(
|
||||
cog_spending("061037123085", years = 2011L, category = "Public Welfare"))
|
||||
sugg <- attr(r, "provenance")$suggestions
|
||||
|
||||
expect_length(sugg, 2L)
|
||||
ids <- vapply(sugg, function(s) s$recipe_id, character(1))
|
||||
expect_setequal(ids, c("welfare_cash_e67_wide", "welfare_cash_e68_wide"))
|
||||
|
||||
e67 <- sugg[[which(ids == "welfare_cash_e67_wide")]]
|
||||
expect_equal(e67$trigger, "suppressed_component")
|
||||
expect_equal(e67$suppressed_amount, 1803872000)
|
||||
expect_equal(e67$suppressed_years, 2011L)
|
||||
expect_equal(e67$suppressed_codes, "E67")
|
||||
expect_equal(e67$hint, "re-run with recipe = 'welfare_cash_e67_wide'")
|
||||
|
||||
e68 <- sugg[[which(ids == "welfare_cash_e68_wide")]]
|
||||
expect_equal(e68$trigger, "suppressed_component")
|
||||
expect_equal(e68$suppressed_amount, 271589000)
|
||||
expect_equal(e68$suppressed_codes, "E68")
|
||||
})
|
||||
|
||||
test_that("uscogdata#9: an empty_year fire keeps its trigger and gains the dollars", {
|
||||
# Corrections is the case that already worked: zero rows in FY2011, so the
|
||||
# row-absence path fires. It must keep firing, keep trigger = "empty_year",
|
||||
# keep its IG counterpart -- and now also report what was suppressed.
|
||||
skip_if_no_corpus()
|
||||
r <- suppressMessages(
|
||||
cog_spending("061037123085", years = 2011L, category = "Corrections"))
|
||||
sugg <- attr(r, "provenance")$suggestions
|
||||
|
||||
expect_length(sugg, 3L)
|
||||
ids <- vapply(sugg, function(s) s$recipe_id, character(1))
|
||||
expect_setequal(ids, c("corrections_combined", "corrections_capital_combined",
|
||||
"corrections_other_capital_combined"))
|
||||
expect_true(all(vapply(sugg, function(s) s$trigger, character(1)) == "empty_year"))
|
||||
|
||||
cc <- sugg[[which(ids == "corrections_combined")]]
|
||||
expect_equal(cc$suppressed_amount, 1371460000)
|
||||
expect_equal(cc$suppressed_codes, "E05")
|
||||
expect_equal(cc$ig_recipe_id, "corrections_ig_local_combined")
|
||||
})
|
||||
|
||||
test_that("uscogdata#9: the revenue verb inherits the same trigger", {
|
||||
# Alaska state FY2011 Miscellaneous Revenue reports $943,842,000 from
|
||||
# U11/U20/U30 while dropping $1,899,995,000 of aggregate-published `U4-`
|
||||
# rents and royalties -- the omission is LARGER than the reported figure.
|
||||
skip_if_no_corpus()
|
||||
r <- suppressMessages(
|
||||
cog_revenue("020000227749", years = 2011L,
|
||||
category = "Miscellaneous Revenue"))
|
||||
sugg <- attr(r, "provenance")$suggestions
|
||||
|
||||
expect_length(sugg, 1L)
|
||||
expect_equal(sugg[[1]]$recipe_id, "rents_royalties_u4_wide")
|
||||
expect_equal(sugg[[1]]$trigger, "suppressed_component")
|
||||
expect_equal(sugg[[1]]$suppressed_amount, 1899995000)
|
||||
expect_equal(sugg[[1]]$suppressed_codes, "U4-")
|
||||
# A revenue recipe must never be handed an M/L expenditure counterpart.
|
||||
expect_null(sugg[[1]]$ig_recipe_id)
|
||||
})
|
||||
|
||||
test_that("I1 + #34: cog_revenue never suggests expenditure-only recipes", {
|
||||
# uscogdata#9 review, finding I1: Corrections is an expenditure-only
|
||||
# category (E04/E05). Before the flow_prefixes fix (#9), .suppressed_components()
|
||||
# measured E04/E05 against cog_revenue()'s OWN view and reported $3.6B as
|
||||
# "suppressed" -- nothing was suppressed at all.
|
||||
#
|
||||
# Issue #34 builds on that: the candidate query now also filters by
|
||||
# category_type ('revenue'), so expenditure-only recipes like corrections_combined
|
||||
# (whose components E04/E05 are classified as 'expenditure' in summary_categories)
|
||||
# are never even considered for a revenue verb. This is stronger than just
|
||||
# suppressing the dollar claim -- it prevents the suggestion from firing at all.
|
||||
skip_if_no_corpus()
|
||||
r <- suppressMessages(
|
||||
cog_revenue("061037123085", years = 2019:2020, category = "Corrections"))
|
||||
sugg <- attr(r, "provenance")$suggestions
|
||||
ids <- vapply(sugg, function(s) s$recipe_id %||% "", character(1))
|
||||
|
||||
# corrections_combined should NOT appear -- its components are expenditure-only.
|
||||
expect_false("corrections_combined" %in% ids)
|
||||
})
|
||||
|
||||
test_that("uscogdata#9: no partial-coverage fire in a modern year", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_spending("061037123085", years = 2019L, category = "Public Welfare")
|
||||
expect_length(attr(r, "provenance")$suggestions, 0L)
|
||||
})
|
||||
|
||||
test_that("uscogdata#9: leaf-and-classified wide-era families never fire", {
|
||||
# higher_ed_e18_wide and general_gov_e89_wide are the control group: their
|
||||
# components (E16/E18, E85/E89) are ordinary classified leaves even in the
|
||||
# wide era, so widening the trigger must leave them silent. This is the
|
||||
# measurement that refutes "it would fire on every category in every legacy
|
||||
# year" -- corpus-wide on the fixture, these two produce zero suppressed rows.
|
||||
skip_if_no_corpus()
|
||||
con <- uscogdata:::.ensure_session()
|
||||
n <- DBI::dbGetQuery(con,
|
||||
"SELECT COUNT(*) AS n
|
||||
FROM long l
|
||||
JOIN harmonization_recipes r
|
||||
ON l.item_code = r.component_code
|
||||
AND l.year BETWEEN r.year_min AND r.year_max
|
||||
WHERE r.recipe_id IN ('higher_ed_e18_wide', 'general_gov_e89_wide')
|
||||
AND l.amt <> 0
|
||||
AND NOT EXISTS (
|
||||
SELECT 1 FROM spending_long_harmonized v
|
||||
WHERE v.canonical_govid = l.canonical_govid
|
||||
AND v.year = l.year AND v.item_code = l.item_code)")$n
|
||||
expect_equal(as.integer(n), 0L)
|
||||
})
|
||||
|
||||
test_that("uscogdata#9: the cli message reports the suppressed dollars", {
|
||||
skip_if_no_corpus()
|
||||
expect_message(
|
||||
cog_spending("061037123085", years = 2011L, category = "Public Welfare"),
|
||||
"1,803,872,000", fixed = TRUE)
|
||||
expect_message(
|
||||
cog_spending("061037123085", years = 2011L, category = "Public Welfare"),
|
||||
"FY2011", fixed = TRUE)
|
||||
expect_message(
|
||||
cog_spending("061037123085", years = 2011L, category = "Public Welfare"),
|
||||
"E67", fixed = TRUE)
|
||||
})
|
||||
|
||||
test_that("uscogdata#9: cog_explain() reports the suppressed dollars", {
|
||||
# cog_explain()'s whole "print" output -- including the Suggestions
|
||||
# section built from cli::cli_ul() -- is emitted on the message stream
|
||||
# (verified empirically 2026-08-04: capture.output(..., type = "output")
|
||||
# returns character(0) for this call; testthat::capture_messages() is what
|
||||
# actually carries it), so that is the stream this test captures.
|
||||
skip_if_no_corpus()
|
||||
r <- suppressMessages(
|
||||
cog_spending("061037123085", years = 2011L, category = "Public Welfare"))
|
||||
out <- paste(testthat::capture_messages(cog_explain(r)), collapse = "")
|
||||
expect_match(out, "271,589,000", fixed = TRUE)
|
||||
})
|
||||
|
||||
test_that("the provenance schema documents the suggestion trigger fields", {
|
||||
sch <- jsonlite::fromJSON(
|
||||
system.file("schemas", "provenance-v1.json", package = "uscogdata"),
|
||||
simplifyVector = FALSE)
|
||||
props <- sch$properties$suggestions$items$properties
|
||||
expect_true(all(c("trigger", "suppressed_amount", "suppressed_years",
|
||||
"suppressed_codes") %in% names(props)))
|
||||
expect_setequal(unlist(props$trigger$enum),
|
||||
c("empty_year", "suppressed_component"))
|
||||
})
|
||||
|
||||
@@ -1,29 +0,0 @@
|
||||
# Mirror of test-spending-pagination.R for cog_revenue(), which shares the
|
||||
# same .verb_spendrev()/.build_verb_sql() pushdown -- see that file for the
|
||||
# incident this fixes.
|
||||
|
||||
test_that("cog_revenue limit/offset page correctly and report total_rows", {
|
||||
skip_if_no_corpus()
|
||||
full <- cog_revenue("121011212191", years = 2019:2020, category = NULL)
|
||||
page <- cog_revenue("121011212191", years = 2019:2020, category = NULL,
|
||||
limit = 5L, offset = 3L)
|
||||
expect_equal(nrow(page), 5L)
|
||||
expect_equal(page[c("year", "canonical_govid", "revenue_subtype", "category")],
|
||||
full[4:8, c("year", "canonical_govid", "revenue_subtype", "category")],
|
||||
ignore_attr = TRUE)
|
||||
expect_equal(attr(page, "total_rows"), nrow(full))
|
||||
})
|
||||
|
||||
test_that("cog_revenue limit unset by default leaves total_rows absent", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_revenue("121011212191", 2020L, "Property Tax")
|
||||
expect_null(attr(r, "total_rows"))
|
||||
})
|
||||
|
||||
test_that("cog_revenue complete + limit conflict aborts the same way as cog_spending", {
|
||||
skip_if_no_corpus()
|
||||
expect_error(
|
||||
cog_revenue("121011212191", 2020L, "Property Tax", complete = TRUE, limit = 5L),
|
||||
class = "uscogdata_complete_pagination_conflict"
|
||||
)
|
||||
})
|
||||
@@ -1,178 +0,0 @@
|
||||
# tests/testthat/test-search-balances-pagination.R
|
||||
#
|
||||
# uscogdata#57. cog_spending()/cog_revenue() gained limit/offset in #39;
|
||||
# cog_gov_search() and cog_balances() did not, so every consumer of those two
|
||||
# was back to materialize-then-slice -- the exact pattern that wedged the
|
||||
# production API for hours on 2026-08-06.
|
||||
#
|
||||
# cog_gov_search() was also the one verb with no LIMIT at all, so an
|
||||
# unfiltered call returns the entire 40,336-row crosswalk by accident.
|
||||
|
||||
# --- cog_gov_search() -------------------------------------------------------
|
||||
|
||||
test_that("cog_gov_search() limit returns the first page of the unpaginated result", {
|
||||
skip_if_no_corpus()
|
||||
full <- cog_gov_search(state = "WI", type = "city")
|
||||
skip_if(nrow(full) < 12L, "fixture has too few WI cities to page")
|
||||
|
||||
page <- cog_gov_search(state = "WI", type = "city", limit = 5L)
|
||||
expect_equal(nrow(page), 5L)
|
||||
expect_equal(page$canonical_govid, full$canonical_govid[1:5])
|
||||
})
|
||||
|
||||
test_that("cog_gov_search() offset skips ahead without gaps or overlap", {
|
||||
skip_if_no_corpus()
|
||||
full <- cog_gov_search(state = "WI", type = "city")
|
||||
skip_if(nrow(full) < 12L, "fixture has too few WI cities to page")
|
||||
|
||||
p1 <- cog_gov_search(state = "WI", type = "city", limit = 5L)
|
||||
p2 <- cog_gov_search(state = "WI", type = "city", limit = 5L, offset = 5L)
|
||||
expect_equal(p2$canonical_govid, full$canonical_govid[6:10])
|
||||
expect_length(intersect(p1$canonical_govid, p2$canonical_govid), 0L)
|
||||
})
|
||||
|
||||
test_that("walking every page reconstructs the unpaginated search exactly", {
|
||||
skip_if_no_corpus()
|
||||
full <- cog_gov_search(state = "WI", type = "city")
|
||||
n <- nrow(full)
|
||||
limit <- 7L
|
||||
pages <- list()
|
||||
offset <- 0L
|
||||
repeat {
|
||||
p <- cog_gov_search(state = "WI", type = "city", limit = limit, offset = offset)
|
||||
if (nrow(p) == 0L) break
|
||||
pages[[length(pages) + 1L]] <- p
|
||||
offset <- offset + limit
|
||||
if (offset > n + limit) stop("test runaway: paging did not terminate")
|
||||
}
|
||||
walked <- dplyr::bind_rows(pages)
|
||||
expect_equal(nrow(walked), n)
|
||||
expect_equal(walked$canonical_govid, full$canonical_govid)
|
||||
})
|
||||
|
||||
test_that("cog_gov_search() total_rows reports the full unpaginated count", {
|
||||
skip_if_no_corpus()
|
||||
full <- cog_gov_search(state = "WI", type = "city")
|
||||
page <- cog_gov_search(state = "WI", type = "city", limit = 3L)
|
||||
expect_equal(attr(page, "total_rows"), nrow(full))
|
||||
})
|
||||
|
||||
test_that("cog_gov_search() offset past the end reports the true total, not zero", {
|
||||
skip_if_no_corpus()
|
||||
full <- cog_gov_search(state = "WI", type = "city")
|
||||
# No row survives to carry COUNT(*) OVER(), so this is the branch that has
|
||||
# to fall back to a second count rather than reporting 0 rows out of 0.
|
||||
page <- cog_gov_search(state = "WI", type = "city",
|
||||
limit = 5L, offset = nrow(full) + 50L)
|
||||
expect_equal(nrow(page), 0L)
|
||||
expect_equal(attr(page, "total_rows"), nrow(full))
|
||||
})
|
||||
|
||||
test_that("cog_gov_search() bounds an otherwise-unfiltered crosswalk sweep", {
|
||||
skip_if_no_corpus()
|
||||
# The reason this verb needed a limit most: with no filter it returns the
|
||||
# whole crosswalk.
|
||||
page <- cog_gov_search(limit = 10L)
|
||||
expect_equal(nrow(page), 10L)
|
||||
expect_gt(attr(page, "total_rows"), 10L)
|
||||
})
|
||||
|
||||
test_that("cog_gov_search() orders by a total order, not population alone", {
|
||||
skip_if_no_corpus()
|
||||
# population_acs is not unique -- NA in particular repeats across many rows
|
||||
# -- so paging on it alone can duplicate a row on one page and drop it from
|
||||
# the next. The tiebreaker is what makes the sequence reproducible.
|
||||
full <- cog_gov_search(state = "WI")
|
||||
skip_if(nrow(full) < 5L, "fixture has too few WI governments")
|
||||
expect_equal(cog_gov_search(state = "WI")$canonical_govid,
|
||||
full$canonical_govid)
|
||||
|
||||
ties <- full[is.na(full$population_acs), ]
|
||||
skip_if(nrow(ties) < 2L, "no tied rows in the fixture to order")
|
||||
expect_false(is.unsorted(ties$canonical_govid))
|
||||
})
|
||||
|
||||
test_that("cog_gov_search() refuses pagination in basket mode", {
|
||||
skip_if_no_corpus()
|
||||
expect_error(
|
||||
cog_gov_search(name = c("MADISON CITY", "MILWAUKEE CITY"),
|
||||
state = c("WI", "WI"), limit = 1L),
|
||||
class = "uscogdata_basket_pagination_conflict"
|
||||
)
|
||||
})
|
||||
|
||||
test_that("cog_gov_search() rejects a malformed limit or offset", {
|
||||
skip_if_no_corpus()
|
||||
expect_error(cog_gov_search(state = "WI", limit = -1L),
|
||||
class = "uscogdata_invalid_pagination")
|
||||
expect_error(cog_gov_search(state = "WI", limit = 5L, offset = -1L),
|
||||
class = "uscogdata_invalid_pagination")
|
||||
})
|
||||
|
||||
# --- cog_balances() ---------------------------------------------------------
|
||||
|
||||
test_that("cog_balances() limit/offset walk the unpaginated result exactly", {
|
||||
skip_if_no_corpus()
|
||||
full <- cog_balances(years = 2019:2020, state = "WI", type = "city")
|
||||
skip_if(nrow(full) < 6L, "fixture has too few WI city balance rows to page")
|
||||
|
||||
key <- c("year", "canonical_govid", "balance_subtype", "amt_nominal")
|
||||
p1 <- cog_balances(years = 2019:2020, state = "WI", type = "city", limit = 3L)
|
||||
p2 <- cog_balances(years = 2019:2020, state = "WI", type = "city",
|
||||
limit = 3L, offset = 3L)
|
||||
|
||||
expect_equal(nrow(p1), 3L)
|
||||
expect_equal(p1[key], full[1:3, key], ignore_attr = TRUE)
|
||||
expect_equal(p2[key], full[4:6, key], ignore_attr = TRUE)
|
||||
# The window-function column is an implementation detail and must not reach
|
||||
# the caller's data frame.
|
||||
expect_false("pagination_total_rows" %in% names(p1))
|
||||
})
|
||||
|
||||
test_that("cog_balances() total_rows reports the full unpaginated count", {
|
||||
skip_if_no_corpus()
|
||||
full <- cog_balances(years = 2019:2020, state = "WI", type = "city")
|
||||
page <- cog_balances(years = 2019:2020, state = "WI", type = "city", limit = 2L)
|
||||
expect_equal(attr(page, "total_rows"), nrow(full))
|
||||
})
|
||||
|
||||
test_that("cog_balances() offset past the end reports the true total", {
|
||||
skip_if_no_corpus()
|
||||
full <- cog_balances(years = 2019:2020, state = "WI", type = "city")
|
||||
page <- cog_balances(years = 2019:2020, state = "WI", type = "city",
|
||||
limit = 5L, offset = nrow(full) + 50L)
|
||||
expect_equal(nrow(page), 0L)
|
||||
expect_equal(attr(page, "total_rows"), nrow(full))
|
||||
})
|
||||
|
||||
test_that("cog_balances() refuses pagination alongside a recipe", {
|
||||
skip_if_no_corpus()
|
||||
expect_error(
|
||||
cog_balances(years = 2011, state = "WI", type = "city",
|
||||
recipe = "cash_securities_z77_wide", limit = 5L),
|
||||
class = "uscogdata_recipe_pagination_conflict"
|
||||
)
|
||||
})
|
||||
|
||||
test_that("cog_balances() rejects a malformed limit or offset", {
|
||||
skip_if_no_corpus()
|
||||
expect_error(cog_balances(years = 2019, state = "WI", type = "city", limit = -1L),
|
||||
class = "uscogdata_invalid_pagination")
|
||||
expect_error(cog_balances(years = 2019, state = "WI", type = "city",
|
||||
limit = 5L, offset = -1L),
|
||||
class = "uscogdata_invalid_pagination")
|
||||
})
|
||||
|
||||
# --- Unchanged without the arguments ----------------------------------------
|
||||
|
||||
test_that("both verbs are unchanged when limit is not supplied", {
|
||||
skip_if_no_corpus()
|
||||
# The adoption contract for cog-api: NULL default, so a formals() probe can
|
||||
# feature-detect without any call site changing behaviour.
|
||||
s <- cog_gov_search(state = "WI", type = "city")
|
||||
b <- cog_balances(years = 2019, state = "WI", type = "city")
|
||||
expect_null(attr(s, "total_rows"))
|
||||
expect_null(attr(b, "total_rows"))
|
||||
expect_true(all(c("limit", "offset") %in% names(formals(cog_gov_search))))
|
||||
expect_true(all(c("limit", "offset") %in% names(formals(cog_balances))))
|
||||
})
|
||||
@@ -1,95 +0,0 @@
|
||||
# cog-api's paginate() used to slice an already-fully-materialized result:
|
||||
# every page of a deep sweep re-ran the whole query and re-listified every
|
||||
# row, just to keep 1000 and discard the rest. For a 193,105-row fleet-wide
|
||||
# query walked 194 pages deep, that repeated the full cost 194 times and
|
||||
# wedged the production server for hours (2026-08-06 incident). limit/offset
|
||||
# here push the slice into the SQL itself, so a page costs O(limit), not
|
||||
# O(full result).
|
||||
|
||||
test_that("limit without offset returns the first page, matching the unpaginated head", {
|
||||
skip_if_no_corpus()
|
||||
full <- cog_spending("121011212191", years = 2019:2020, category = NULL)
|
||||
page <- cog_spending("121011212191", years = 2019:2020, category = NULL,
|
||||
limit = 10L)
|
||||
expect_equal(nrow(page), 10L)
|
||||
expect_equal(page[c("year", "canonical_govid", "spend_subtype", "category")],
|
||||
full[1:10, c("year", "canonical_govid", "spend_subtype", "category")],
|
||||
ignore_attr = TRUE)
|
||||
})
|
||||
|
||||
test_that("offset skips ahead without gaps or overlap", {
|
||||
skip_if_no_corpus()
|
||||
full <- cog_spending("121011212191", years = 2019:2020, category = NULL)
|
||||
page2 <- cog_spending("121011212191", years = 2019:2020, category = NULL,
|
||||
limit = 10L, offset = 10L)
|
||||
expect_equal(nrow(page2), 10L)
|
||||
expect_equal(page2[c("year", "canonical_govid", "spend_subtype", "category")],
|
||||
full[11:20, c("year", "canonical_govid", "spend_subtype", "category")],
|
||||
ignore_attr = TRUE)
|
||||
})
|
||||
|
||||
test_that("walking every page reconstructs the unpaginated result exactly", {
|
||||
skip_if_no_corpus()
|
||||
full <- cog_spending("121011212191", years = 2019:2020, category = NULL)
|
||||
n <- nrow(full)
|
||||
limit <- 7L
|
||||
pages <- list()
|
||||
offset <- 0L
|
||||
repeat {
|
||||
p <- cog_spending("121011212191", years = 2019:2020, category = NULL,
|
||||
limit = limit, offset = offset)
|
||||
if (nrow(p) == 0L) break
|
||||
pages[[length(pages) + 1L]] <- p
|
||||
offset <- offset + limit
|
||||
if (offset > n + limit) stop("test runaway: paging did not terminate")
|
||||
}
|
||||
walked <- dplyr::bind_rows(pages)
|
||||
expect_equal(nrow(walked), n)
|
||||
key_cols <- c("year", "canonical_govid", "spend_subtype", "category", "amt_nominal")
|
||||
expect_equal(walked[key_cols], full[key_cols], ignore_attr = TRUE)
|
||||
})
|
||||
|
||||
test_that("total_rows attribute reports the full unpaginated count", {
|
||||
skip_if_no_corpus()
|
||||
full <- cog_spending("121011212191", years = 2019:2020, category = NULL)
|
||||
page <- cog_spending("121011212191", years = 2019:2020, category = NULL,
|
||||
limit = 5L, offset = 0L)
|
||||
expect_equal(attr(page, "total_rows"), nrow(full))
|
||||
})
|
||||
|
||||
test_that("offset past the end returns zero rows, not an error", {
|
||||
skip_if_no_corpus()
|
||||
full <- cog_spending("121011212191", years = 2019:2020, category = NULL)
|
||||
page <- cog_spending("121011212191", years = 2019:2020, category = NULL,
|
||||
limit = 10L, offset = nrow(full) + 100L)
|
||||
expect_equal(nrow(page), 0L)
|
||||
expect_equal(attr(page, "total_rows"), nrow(full))
|
||||
})
|
||||
|
||||
test_that("limit is unset by default -- unpaginated calls are unaffected", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_spending("121011212191", 2020L, "Corrections")
|
||||
expect_null(attr(r, "total_rows"))
|
||||
})
|
||||
|
||||
test_that("per_capita and adjust_to_year still apply correctly within a page", {
|
||||
skip_if_no_corpus()
|
||||
full <- cog_spending("121011212191", years = 2020L, category = NULL,
|
||||
per_capita = TRUE, adjust_to_year = 2022L)
|
||||
page <- cog_spending("121011212191", years = 2020L, category = NULL,
|
||||
per_capita = TRUE, adjust_to_year = 2022L,
|
||||
limit = 3L, offset = 2L)
|
||||
expect_equal(page[c("amt_nominal", "amt_real", "amt_per_capita_nominal",
|
||||
"amt_per_capita_real")],
|
||||
full[3:5, c("amt_nominal", "amt_real", "amt_per_capita_nominal",
|
||||
"amt_per_capita_real")],
|
||||
ignore_attr = TRUE)
|
||||
})
|
||||
|
||||
test_that("complete = TRUE with limit aborts -- pagination over a partial grid is undefined", {
|
||||
skip_if_no_corpus()
|
||||
expect_error(
|
||||
cog_spending("121011212191", 2020L, "Corrections", complete = TRUE, limit = 5L),
|
||||
class = "uscogdata_complete_pagination_conflict"
|
||||
)
|
||||
})
|
||||
@@ -103,7 +103,7 @@ test_that("inst/sql/22- and 23- harmonized views enforce every WHERE predicate (
|
||||
sql_dir <- system.file("sql", package = "uscogdata")
|
||||
.read_view_sql <- function(filename) {
|
||||
txt <- paste(readLines(file.path(sql_dir, filename), warn = FALSE), collapse = "\n")
|
||||
uscogdata:::.render_view_sql(txt, paste0(tmp, "/"))
|
||||
gsub("\\{url\\}", paste0(tmp, "/"), txt, fixed = FALSE)
|
||||
}
|
||||
|
||||
con <- DBI::dbConnect(duckdb::duckdb())
|
||||
@@ -183,7 +183,7 @@ test_that("inst/sql/24- and 25- IG views retain aggregates, COALESCE NULL harmon
|
||||
sql_dir <- system.file("sql", package = "uscogdata")
|
||||
.read_view_sql <- function(filename) {
|
||||
txt <- paste(readLines(file.path(sql_dir, filename), warn = FALSE), collapse = "\n")
|
||||
uscogdata:::.render_view_sql(txt, paste0(tmp, "/"))
|
||||
gsub("\\{url\\}", paste0(tmp, "/"), txt, fixed = FALSE)
|
||||
}
|
||||
|
||||
con <- DBI::dbConnect(duckdb::duckdb())
|
||||
@@ -335,7 +335,7 @@ test_that(".harmonization_view_files guard is necessary: registration against a
|
||||
sql_dir <- system.file("sql", package = "uscogdata")
|
||||
.read_view_sql <- function(filename) {
|
||||
txt <- paste(readLines(file.path(sql_dir, filename), warn = FALSE), collapse = "\n")
|
||||
uscogdata:::.render_view_sql(txt, url)
|
||||
gsub("\\{url\\}", url, txt, fixed = FALSE)
|
||||
}
|
||||
con2 <- DBI::dbConnect(duckdb::duckdb())
|
||||
on.exit(DBI::dbDisconnect(con2, shutdown = TRUE), add = TRUE)
|
||||
|
||||
Reference in New Issue
Block a user