Compare commits
162
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f4ab9b6d90
|
||
|
|
9508b98676 | ||
|
|
0fbae00e27
|
||
|
|
698812a25c | ||
|
|
d74ecdd4a5 | ||
|
|
72b2cc3a26
|
||
|
|
28c500f47a
|
||
|
|
fe9238a6ef | ||
|
|
6392a74013
|
||
|
|
de2ba0cfb9 | ||
|
|
e912a926c2 | ||
|
|
44e4953f94
|
||
|
|
a5500f0b6a
|
||
|
|
da2839f885
|
||
|
|
331399ab86
|
||
|
|
5582c6cb57
|
||
|
|
013af5b0d2
|
||
|
|
4a92f36d44
|
||
|
|
a67735f121
|
||
|
|
b9f7f7d8d3
|
||
|
|
4300b636b1
|
||
|
|
99e1e86e37
|
||
|
|
c042ee0b90
|
||
|
|
29dc8199e0
|
||
|
|
619b167ab1
|
||
|
|
dfda39051e
|
||
|
|
785f3af16d
|
||
|
|
c587c8ba87 | ||
|
|
6a06302036
|
||
|
|
e3ab26c3e6 | ||
|
|
498950afa6
|
||
|
|
a5f86d87b3
|
||
|
|
61b9c95731
|
||
|
|
44e9b40b86
|
||
|
|
f1e9aa383a
|
||
|
|
12a9be110f
|
||
|
|
503fa6562f
|
||
|
|
11ae99c382
|
||
|
|
5e22e940e7
|
||
|
|
2fc9e7585b | ||
|
|
8bf9c4ccc1
|
||
|
|
77074621d8
|
||
|
|
4b749205a5
|
||
|
|
f77adb6c83
|
||
|
|
230f3401c4
|
||
|
|
7522b48a08
|
||
|
|
693f8d81a6
|
||
|
|
db35fa9058
|
||
|
|
cabe2e2799
|
||
|
|
6cd219a291 | ||
|
|
e067a5930f
|
||
|
|
342debaefa
|
||
|
|
b59b79b2d5 | ||
|
|
5668d6b102
|
||
|
|
0a6d878a36 | ||
|
|
da726a61f6
|
||
|
|
03c313b46d | ||
|
|
2c532bde19
|
||
|
|
a9e80858d4
|
||
|
|
fde62eb6cc
|
||
|
|
22c2478634
|
||
|
|
225cd60968
|
||
|
|
b03f095e49
|
||
|
|
724b6bd58b
|
||
|
|
82e4face4e
|
||
|
|
6c5bdb3048
|
||
|
|
b8189aeb7f
|
||
|
|
90d2e6019e
|
||
|
|
769164c824
|
||
|
|
de3a58d105
|
||
|
|
cdb574d3d0
|
||
|
|
a281a9621f
|
||
|
|
825ac394f2
|
||
|
|
d09bfd6aef
|
||
|
|
a11e29a0e0
|
||
|
|
7ac4dc6882
|
||
|
|
57212e3399
|
||
|
|
9f9d40e1c3
|
||
|
|
d7e14156ff
|
||
|
|
de7ccbebc7 | ||
|
|
4b23dbd9f4
|
||
|
|
93300ae0c1
|
||
|
|
5d77d39711 | ||
|
|
7d798b9937
|
||
|
|
915a4d0678 | ||
|
|
6f98d061a9 | ||
|
|
d95c9032c5
|
||
|
|
af85a23ea7
|
||
|
|
8db944e4a0 | ||
|
|
2e8383b098
|
||
|
|
d006dea6e4
|
||
|
|
ebac39e6de | ||
|
|
47dc08c4b0 | ||
|
|
1d553a788f
|
||
|
|
c375c55da7
|
||
|
|
82acda6f93 | ||
|
|
9233c3d18e
|
||
|
|
1f257812b6 | ||
|
|
d258cef8c5 | ||
|
|
e7d3a7a310 | ||
|
|
a4eb80d823 | ||
|
|
aba7ffbac2 | ||
|
|
c1c6b5a6ba | ||
|
|
c7260cb20c | ||
|
|
54ece11867 | ||
|
|
3bd9b1f011 | ||
|
|
24b2ff7d8c | ||
|
|
c28712f62f | ||
|
|
7913b0f664 | ||
|
|
887acf7e81 | ||
|
|
81fd1a5279 | ||
|
|
e2088458e1 | ||
|
|
fefd4fe969 | ||
|
|
7ed1da9b79 | ||
|
|
9240a18ea3 | ||
|
|
46fed3a241 | ||
|
|
c9d1a05d4f | ||
|
|
fcecd62a03 | ||
|
|
e581e7360c | ||
|
|
748ca4a56e | ||
|
|
fa40266d07 | ||
|
|
3583c05852
|
||
|
|
bd53230ae7 | ||
|
|
e813ffd3aa | ||
|
|
b0df1ec668 | ||
|
|
77f48047b1
|
||
|
|
4de915b557
|
||
|
|
7818cd2b1a
|
||
|
|
3b725770d2 | ||
|
|
4d61692f05
|
||
|
|
70cf553828
|
||
|
|
3c55447308
|
||
|
|
92c9a7382e
|
||
|
|
570a9408a2
|
||
|
|
e635a1fc9e
|
||
|
|
874347242b
|
||
|
|
0dd3f15ada | ||
|
|
919548685b | ||
|
|
716cfe25e5 | ||
|
|
33c0274727 | ||
|
|
c46354f049
|
||
|
|
916212c327
|
||
|
|
a2ced368f5
|
||
|
|
b7ebb4cd88
|
||
|
|
c334be7706
|
||
|
|
cadce8d528 | ||
|
|
a92450ff76
|
||
|
|
54dd40a61d
|
||
|
|
807ed35cb7
|
||
|
|
9ae46746c0 | ||
|
|
9238b04b69
|
||
|
|
cbc867bed1 | ||
|
|
4ea0583d3a
|
||
|
|
e4a105013e | ||
|
|
21b3d66c0e
|
||
|
|
df3fe3731b | ||
|
|
ed9658d267
|
||
|
|
a28fb2e19b | ||
|
|
24e4449be7 | ||
|
|
cfcda04e0c | ||
|
|
a25ba5f348 | ||
|
|
e7fa51eec7 |
+8
-1
@@ -3,10 +3,17 @@
|
|||||||
^\.Rproj\.user$
|
^\.Rproj\.user$
|
||||||
^_pkgdown\.yml$
|
^_pkgdown\.yml$
|
||||||
^docs$
|
^docs$
|
||||||
|
^Meta$
|
||||||
|
^doc$
|
||||||
^pkgdown$
|
^pkgdown$
|
||||||
^\.github$
|
^\.github$
|
||||||
^LICENSE\.md$
|
^LICENSE\.md$
|
||||||
^\.git$
|
^\.git$
|
||||||
^\.gitignore$
|
^\.gitignore$
|
||||||
\.gitkeep$
|
\.gitkeep$
|
||||||
^vignettes$
|
^specs$
|
||||||
|
^plans$
|
||||||
|
^\.gitea$
|
||||||
|
^CLAUDE\.md$
|
||||||
|
^\.superpowers$
|
||||||
|
^CONTRIBUTING\.md$
|
||||||
|
|||||||
@@ -11,6 +11,21 @@ jobs:
|
|||||||
steps:
|
steps:
|
||||||
- name: Install system libraries and Node.js (required by actions/checkout)
|
- name: Install system libraries and Node.js (required by actions/checkout)
|
||||||
run: |
|
run: |
|
||||||
|
# Switch apt to HTTPS mirrors. Measured from this runner on
|
||||||
|
# 2026-08-04: the SAME index file takes 20.1s over http:// and 3.1s
|
||||||
|
# over https://. apt fetches many indexes serially, so http:// does
|
||||||
|
# not read as "slow" -- it reads as a hang (zero bytes in
|
||||||
|
# /var/cache/apt/archives after 3+ minutes, apt's http workers parked
|
||||||
|
# in S state). rocker/r-ver:4.4 already ships ca-certificates and
|
||||||
|
# apt 2.8.3 has the https method built in, so nothing needs to be
|
||||||
|
# installed over http first to bootstrap this.
|
||||||
|
# `|| true` because the step runs under `sh -e`: on an image whose
|
||||||
|
# sources live in the other location, the missing-file sed must not
|
||||||
|
# kill the job.
|
||||||
|
sed -i -E 's#http://(archive|security)\.ubuntu\.com#https://\1.ubuntu.com#g' \
|
||||||
|
/etc/apt/sources.list.d/ubuntu.sources 2>/dev/null || true
|
||||||
|
sed -i -E 's#http://(archive|security)\.ubuntu\.com#https://\1.ubuntu.com#g' \
|
||||||
|
/etc/apt/sources.list 2>/dev/null || true
|
||||||
apt-get update -qq
|
apt-get update -qq
|
||||||
apt-get install -y --no-install-recommends \
|
apt-get install -y --no-install-recommends \
|
||||||
nodejs git \
|
nodejs git \
|
||||||
|
|||||||
@@ -0,0 +1,43 @@
|
|||||||
|
# Mirror the canonical Gitea repo to the public GitHub mirror.
|
||||||
|
#
|
||||||
|
# Deliberately a plain `git push`, NOT Gitea's built-in push mirror. A push
|
||||||
|
# mirror force-updates the refs it owns: if anyone ever clicks Merge on a
|
||||||
|
# GitHub PR, the next sync silently overwrites main, the PR still displays
|
||||||
|
# "Merged", the commit becomes unreachable, and nothing anywhere says so.
|
||||||
|
# A non-force push is REJECTED as non-fast-forward the moment that happens,
|
||||||
|
# turning a silent data-loss trap into a red CI run in a place we already look.
|
||||||
|
#
|
||||||
|
# Do NOT add --force here, and do NOT add GitHub branch protection to the
|
||||||
|
# mirror: protection rules block the mirror's legitimate pushes too, breaking
|
||||||
|
# normal syncing to catch an abnormal case.
|
||||||
|
#
|
||||||
|
# PAT_GH is a GitHub personal access token (repo + workflow scope; workflow is
|
||||||
|
# required because this pushes .github/workflows/). It is stored as a Gitea
|
||||||
|
# Actions secret. The name cannot begin with GITHUB_ or GITEA_ -- Gitea
|
||||||
|
# reserves both prefixes for its own injected variables and rejects the secret.
|
||||||
|
name: Mirror to GitHub
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches: [main]
|
||||||
|
tags: ['v*']
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
mirror:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
|
|
||||||
|
- name: Push main and tags to the GitHub mirror
|
||||||
|
env:
|
||||||
|
PAT_GH: ${{ secrets.PAT_GH }}
|
||||||
|
run: |
|
||||||
|
set -eu
|
||||||
|
if [ -z "${PAT_GH:-}" ]; then
|
||||||
|
echo "PAT_GH is unset -- add it under Settings > Actions > Secrets." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
git push "https://x-access-token:${PAT_GH}@github.com/civilytics/uscogdata.git" \
|
||||||
|
HEAD:refs/heads/main --tags
|
||||||
@@ -0,0 +1,61 @@
|
|||||||
|
# Multi-platform R CMD check, running on the GitHub mirror.
|
||||||
|
#
|
||||||
|
# This exists because the canonical Gitea runner is Linux-only, and this
|
||||||
|
# package hard-depends on duckdb and httr2 -- both compiled, both with real
|
||||||
|
# platform variance -- while having never been checked on Windows or macOS.
|
||||||
|
# A large share of the audience is on Windows.
|
||||||
|
#
|
||||||
|
# Gitea reads .gitea/workflows and GitHub reads .github/workflows, so this
|
||||||
|
# file is inert on the canonical repo and coexists with the Gitea CI that
|
||||||
|
# remains authoritative for deploys.
|
||||||
|
#
|
||||||
|
# The suite needs NO credentials: tests/testthat/setup.R points USCOGDATA_URL
|
||||||
|
# at the bundled fixture corpus. That is exactly why inst/extdata/fixture_corpus
|
||||||
|
# must never be added to .Rbuildignore.
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches: [main]
|
||||||
|
pull_request:
|
||||||
|
|
||||||
|
name: R-CMD-check
|
||||||
|
|
||||||
|
permissions: read-all
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
R-CMD-check:
|
||||||
|
runs-on: ${{ matrix.config.os }}
|
||||||
|
name: ${{ matrix.config.os }} (${{ matrix.config.r }})
|
||||||
|
|
||||||
|
strategy:
|
||||||
|
fail-fast: false
|
||||||
|
matrix:
|
||||||
|
config:
|
||||||
|
- {os: macos-latest, r: 'release'}
|
||||||
|
- {os: windows-latest, r: 'release'}
|
||||||
|
- {os: ubuntu-latest, r: 'devel', http-user-agent: 'release'}
|
||||||
|
- {os: ubuntu-latest, r: 'release'}
|
||||||
|
|
||||||
|
env:
|
||||||
|
GITHUB_PAT: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
R_KEEP_PKG_SOURCE: yes
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- uses: r-lib/actions/setup-pandoc@v2
|
||||||
|
|
||||||
|
- uses: r-lib/actions/setup-r@v2
|
||||||
|
with:
|
||||||
|
r-version: ${{ matrix.config.r }}
|
||||||
|
http-user-agent: ${{ matrix.config.http-user-agent }}
|
||||||
|
use-public-rspm: true
|
||||||
|
|
||||||
|
- uses: r-lib/actions/setup-r-dependencies@v2
|
||||||
|
with:
|
||||||
|
extra-packages: any::rcmdcheck
|
||||||
|
needs: check
|
||||||
|
|
||||||
|
- uses: r-lib/actions/check-r-package@v2
|
||||||
|
with:
|
||||||
|
upload-snapshots: true
|
||||||
|
build_args: 'c("--no-manual")'
|
||||||
@@ -9,3 +9,6 @@ docs/
|
|||||||
/Meta/
|
/Meta/
|
||||||
.DS_Store
|
.DS_Store
|
||||||
/.quarto/
|
/.quarto/
|
||||||
|
|
||||||
|
# SDD working artifacts (ledger, briefs, review packages) — plans/ stays tracked
|
||||||
|
.superpowers/sdd/
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -28,8 +28,23 @@ USCOGDATA_URL (local path or https://)
|
|||||||
- `R/session.R` — `cog_open()`, `cog_close()`, `.ensure_session()`, `.coerce_govid_input()`
|
- `R/session.R` — `cog_open()`, `cog_close()`, `.ensure_session()`, `.coerce_govid_input()`
|
||||||
- `R/manifest.R` — `.fetch_or_cache_manifest()`, `.is_local_path()` (local paths bypass HTTP/cache)
|
- `R/manifest.R` — `.fetch_or_cache_manifest()`, `.is_local_path()` (local paths bypass HTTP/cache)
|
||||||
- `R/views.R` — `.register_views()` (substitutes `{url}` into SQL files at `inst/sql/`)
|
- `R/views.R` — `.register_views()` (substitutes `{url}` into SQL files at `inst/sql/`)
|
||||||
- `inst/sql/` — 7 SQL view definitions: `long`, `spending_long`, `revenue_long`, `canonical_fips_xwalk`, `summary_categories`, `spending_annotated`, `revenue_annotated`
|
- `inst/sql/` — **23** SQL view definitions (measured), numbered by load order
|
||||||
|
(`10-` through `46-`): the `*_long` layer (`long`, `spending_long`,
|
||||||
|
`revenue_long`, `ig_long`, `balance_long`, plus `_harmonized` variants of
|
||||||
|
`spending_long`/`revenue_long`/`ig_long`), the `*_annotated` layer
|
||||||
|
(`spending_annotated`, `revenue_annotated`, `ig_annotated`,
|
||||||
|
`balance_annotated`, plus `_harmonized` variants of `spending_annotated`/
|
||||||
|
`revenue_annotated`/`ig_annotated`), and metadata views
|
||||||
|
(`canonical_fips_xwalk`, `summary_categories`, `gov_population_yearly`,
|
||||||
|
`harmonization_map`, `harmonization_recipes`, `series_breaks_pq`,
|
||||||
|
`representation`, `code_set`)
|
||||||
- `R/spending.R` / `R/revenue.R` — `cog_spending()` / `cog_revenue()` via shared `.verb_spendrev()`
|
- `R/spending.R` / `R/revenue.R` — `cog_spending()` / `cog_revenue()` via shared `.verb_spendrev()`
|
||||||
|
- `R/balances.R` — `cog_balances()`. A third money-adjacent verb, but returns a
|
||||||
|
**stock** (a balance at a point in time) rather than a **flow** (activity
|
||||||
|
over a fiscal year), so it does NOT route through `.verb_spendrev()` and has
|
||||||
|
no `expenditure_concept`/`revenue_concept`/`complete`/`subtype` arguments.
|
||||||
|
`R/balance_caveats.R` attaches `provenance$balance_caveats` (GAAP-vs-gross
|
||||||
|
disclosure + measured per-subtype coverage windows).
|
||||||
- `R/rollup.R` — `cog_geographic_rollup()` (accepts named list of govids by layer)
|
- `R/rollup.R` — `cog_geographic_rollup()` (accepts named list of govids by layer)
|
||||||
- `R/peers.R` — `cog_find_peers()` + `cog_peer_compare()`
|
- `R/peers.R` — `cog_find_peers()` + `cog_peer_compare()`
|
||||||
- `R/search.R` — `cog_gov_search()` (name pattern, state, type filters)
|
- `R/search.R` — `cog_gov_search()` (name pattern, state, type filters)
|
||||||
@@ -45,28 +60,31 @@ USCOGDATA_URL (local path or https://)
|
|||||||
Any value without `://` is treated as a local path by `.is_local_path()` and reads
|
Any value without `://` is treated as a local path by `.is_local_path()` and reads
|
||||||
`manifest.json` directly from disk (no HTTP, no TTL cache).
|
`manifest.json` directly from disk (no HTTP, no TTL cache).
|
||||||
|
|
||||||
## Current State (2026-04-27)
|
## Current State (2026-08-03)
|
||||||
|
|
||||||
**Version:** 0.1.0 (pre-release)
|
**Version:** 0.1.0 (pre-release)
|
||||||
**Branch:** `main`, commit `d65e9fe`
|
**Branch:** `feat/cog-balances-25`, commit `fde62eb`
|
||||||
**Tests:** 181 PASS / 0 FAIL / 0 SKIP
|
**Tests:** 788 PASS / 0 FAIL / 0 SKIP / 0 WARN (measured `testthat::test_local()`, 2026-08-03, after the final-review fix wave)
|
||||||
**CI:** Gitea Actions green (`.gitea/workflows/ci.yml`)
|
**CI:** Gitea Actions green (`.gitea/workflows/ci.yml`)
|
||||||
|
|
||||||
### Completed (Tasks 2.1–2.7)
|
### Completed (Tasks 2.1–2.7)
|
||||||
|
|
||||||
All 8 exported verbs implemented and tested:
|
All **14** exports implemented and tested (measured from `NAMESPACE`):
|
||||||
`cog_spending`, `cog_revenue`, `cog_explain`, `cog_geographic_rollup`,
|
`cog_spending`, `cog_revenue`, `cog_balances`, `cog_explain`,
|
||||||
`cog_find_peers`, `cog_peer_compare`, `cog_gov_search`, `cog_mirror`,
|
`cog_geographic_rollup`, `cog_find_peers`, `cog_peer_compare`,
|
||||||
plus `cog_categories`.
|
`cog_gov_search`, `cog_mirror`, `cog_categories`, `cog_recipes`,
|
||||||
|
`cog_manifest`, `cog_basket_resolution`, `cog_basket_unresolved`.
|
||||||
|
|
||||||
Bundled fixture corpus at `inst/extdata/fixture_corpus/` (3.6 MB, years
|
Bundled fixture corpus at `inst/extdata/fixture_corpus/` (years
|
||||||
2019+2020, all 50 states). Tests run fully offline — no credentials needed.
|
2011, 2012, 2019, 2020 — measured via DuckDB `read_parquet(hive_partitioning=1)`,
|
||||||
|
2026-08-03; all 50 states). Tests run fully offline — no credentials needed.
|
||||||
|
|
||||||
### Remaining to v0.1 release
|
### Remaining to v0.1 release
|
||||||
|
|
||||||
1. **Task 2.8 — Docs:** roxygen `@param`/`@return`/`@examples` on all exports;
|
1. **Task 2.8 — Docs:** mostly done — all 14 exports have a `man/*.Rd`,
|
||||||
full `README.md`; `_pkgdown.yml`; `devtools::document()` + `pkgdown::build_site()`.
|
`README.md` and `_pkgdown.yml` exist, and `vignettes/` carries
|
||||||
Vignettes can be stubbed for v0.1.
|
`total-spending.Rmd` + `population-denominators.Rmd`. Outstanding:
|
||||||
|
`pkgdown::build_site()` has never been run (no `docs/`).
|
||||||
|
|
||||||
2. **Phase 3 — cog_explorer bridge:** create
|
2. **Phase 3 — cog_explorer bridge:** create
|
||||||
`cog_explorer/examples/hello_world_uscogdata.Rmd` (installs from Gitea, runs
|
`cog_explorer/examples/hello_world_uscogdata.Rmd` (installs from Gitea, runs
|
||||||
@@ -99,6 +117,17 @@ devtools::test()
|
|||||||
- All verbs call `.ensure_session()` first, then query via `DBI::dbGetQuery()`
|
- All verbs call `.ensure_session()` first, then query via `DBI::dbGetQuery()`
|
||||||
- Return value is always a `tbl_df` with a `provenance` attribute
|
- Return value is always a `tbl_df` with a `provenance` attribute
|
||||||
- govid inputs always go through `.coerce_govid_input()` (accepts character or data frame)
|
- govid inputs always go through `.coerce_govid_input()` (accepts character or data frame)
|
||||||
- SQL lives in `inst/sql/` — never inline SQL strings in R files
|
- SQL has two layers. **View definitions** live in `inst/sql/` and are
|
||||||
|
registered by `.register_views()`, which globs the directory in sorted order
|
||||||
|
and substitutes `{url}`. **Query construction** is inline `sprintf()` in R
|
||||||
|
(`.build_verb_sql()`, `.run_recipe()`, `.attach_per_capita()`). Add a view as
|
||||||
|
a numbered `.sql` file; build a query in R.
|
||||||
- No arrow dependency — DuckDB reads parquet natively
|
- No arrow dependency — DuckDB reads parquet natively
|
||||||
- `withr` is a Suggests-only dep; only used in tests
|
- `withr` is a Suggests-only dep; only used in tests
|
||||||
|
|
||||||
|
## Domain context — read this first
|
||||||
|
|
||||||
|
**Before doing any work in this repo, read `~/.claude/memory/values/civilytics.md`.**
|
||||||
|
It carries the purpose, direction, and constraints for this domain. It is not optional
|
||||||
|
context — read it before planning or writing code, not after. (An `@` import will not
|
||||||
|
work here; project-level imports don't preload. The read is the mechanism.)
|
||||||
|
|||||||
+105
@@ -0,0 +1,105 @@
|
|||||||
|
# Contributing to uscogdata
|
||||||
|
|
||||||
|
Thanks for reading this — a package like this gets better mostly through people
|
||||||
|
noticing that a number looks wrong.
|
||||||
|
|
||||||
|
## Where the code lives
|
||||||
|
|
||||||
|
Development happens on **Gitea**, at
|
||||||
|
`gitea.civilytics.org/Civilytics/uscogdata`. The repository at
|
||||||
|
`github.com/civilytics/uscogdata` is a **mirror** that accepts issues and pull
|
||||||
|
requests.
|
||||||
|
|
||||||
|
## What happens to a GitHub pull request
|
||||||
|
|
||||||
|
Open it normally. Behind the scenes it is fetched and landed on the canonical
|
||||||
|
Gitea repository, then syncs back:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
git fetch github refs/pull/42/head:pr-42
|
||||||
|
git switch main && git merge --no-ff pr-42
|
||||||
|
git push origin main # Gitea -> mirror -> GitHub
|
||||||
|
```
|
||||||
|
|
||||||
|
Because the merge preserves your commits at their original SHAs, **GitHub marks
|
||||||
|
your PR merged on its own** as soon as the mirror syncs. So:
|
||||||
|
|
||||||
|
> If your pull request closes as "Merged" without anyone visibly clicking
|
||||||
|
> Merge, that is the normal, successful outcome — not a rejection.
|
||||||
|
|
||||||
|
Substantial contributions get a `ctb` entry in `DESCRIPTION`, which surfaces in
|
||||||
|
`citation("uscogdata")`.
|
||||||
|
|
||||||
|
There is no CLA and no DCO sign-off requirement.
|
||||||
|
|
||||||
|
## Running the tests
|
||||||
|
|
||||||
|
```r
|
||||||
|
devtools::test() # bundled fixture; no network, no credentials
|
||||||
|
```
|
||||||
|
|
||||||
|
`tests/testthat/setup.R` points `USCOGDATA_URL` at
|
||||||
|
`inst/extdata/fixture_corpus/` automatically — a four-year slice (2011, 2012,
|
||||||
|
2019, 2020) covering all 50 states. That is the whole data setup.
|
||||||
|
|
||||||
|
## Testing against the live corpus
|
||||||
|
|
||||||
|
```sh
|
||||||
|
USCOGDATA_LIVE_TEST=true Rscript -e 'devtools::test(filter = "live-corpus")'
|
||||||
|
```
|
||||||
|
|
||||||
|
This is worth understanding rather than skipping. Until 0.3.0 the package
|
||||||
|
**could not read a remote corpus at all** — the partitioned view used a glob,
|
||||||
|
and DuckDB cannot expand a glob over generic HTTP. It went unnoticed for months
|
||||||
|
because every test path used a local corpus (the bundled fixture), and so did
|
||||||
|
the production API (a host mount). Nothing exercised the package the way a new
|
||||||
|
user does.
|
||||||
|
|
||||||
|
`test-live-corpus.R` is the only test that runs with no `USCOGDATA_URL`, no
|
||||||
|
option, and no fixture. If you change anything touching view registration,
|
||||||
|
manifest handling, or configuration, run it.
|
||||||
|
|
||||||
|
## Do not exclude the fixture from the build
|
||||||
|
|
||||||
|
There is a temptation to add `^inst/extdata/fixture_corpus$` to
|
||||||
|
`.Rbuildignore` because 15 MB feels large for a package. Don't:
|
||||||
|
|
||||||
|
- `vignette("total-spending")` reads from it and would fail to build.
|
||||||
|
- `R CMD check` on r-universe and GitHub Actions would have no corpus, so the
|
||||||
|
suite could not run without credentials.
|
||||||
|
|
||||||
|
This package is not going to CRAN, so its 5 MB guidance does not apply. A
|
||||||
|
package-size NOTE in `R CMD check` is expected and acceptable.
|
||||||
|
|
||||||
|
## Downstream consumers
|
||||||
|
|
||||||
|
`cog-api` depends on this package and its CI clones uscogdata at
|
||||||
|
`USCOGDATA_REF`, **defaulting to `main`**. There is no pin. Anything merged
|
||||||
|
here reaches the API's next build, so before merging a change to the reader,
|
||||||
|
run the API suite against your branch:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
Rscript -e "remotes::install_local('/path/to/uscogdata', upgrade = 'never')"
|
||||||
|
cd /path/to/cog-api/api/tests/testthat
|
||||||
|
Rscript -e 'testthat::test_dir(".", stop_on_failure = TRUE)'
|
||||||
|
```
|
||||||
|
|
||||||
|
The API calls only exported verbs, so internal refactors are usually safe —
|
||||||
|
but "usually" is not a release gate.
|
||||||
|
|
||||||
|
## Release checklist
|
||||||
|
|
||||||
|
1. `devtools::test()` — green against the bundled fixture, offline.
|
||||||
|
2. `USCOGDATA_LIVE_TEST=true devtools::test()` — green against the live corpus.
|
||||||
|
3. cog-api suite green against this branch (above).
|
||||||
|
4. `devtools::check(args = "--as-cran")` — 0 errors, 0 warnings.
|
||||||
|
5. `pkgdown::build_site()` completes.
|
||||||
|
6. Vignettes resolve from an installed copy:
|
||||||
|
`vignette("total-spending", package = "uscogdata")`.
|
||||||
|
7. **Cold-start check**: on a machine that has never had this package,
|
||||||
|
install it and run the README quickstart verbatim with no environment
|
||||||
|
variables set. This is the only check that catches a
|
||||||
|
corpus-unreachable defect, and its absence is why 0.3.0 needed fixing.
|
||||||
|
8. Bump `Version` and add a `NEWS.md` section.
|
||||||
|
9. Tag, then update the r-universe registry pin at
|
||||||
|
`github.com/civilytics/civilytics.r-universe.dev`.
|
||||||
+12
-5
@@ -1,14 +1,21 @@
|
|||||||
Package: uscogdata
|
Package: uscogdata
|
||||||
Type: Package
|
Type: Package
|
||||||
Title: Curated Reader for the Civilytics US Census of Governments Finance Corpus
|
Title: Curated Reader for the Civilytics US Census of Governments Finance Corpus
|
||||||
Version: 0.1.0
|
Version: 0.4.0
|
||||||
Authors@R:
|
Authors@R: c(
|
||||||
person("Civilytics", , , "jknowles@gmail.com", role = c("aut", "cre"))
|
person(c("Jared", "E."), "Knowles",
|
||||||
|
email = "jared@civilytics.com",
|
||||||
|
role = c("aut", "cre"),
|
||||||
|
comment = c(ORCID = "0000-0003-0005-9478")),
|
||||||
|
person("Civilytics Consulting LLC", role = c("cph", "fnd")))
|
||||||
Description: Curated R verbs over the Civilytics US Census of Governments
|
Description: Curated R verbs over the Civilytics US Census of Governments
|
||||||
finance corpus. Provides unit-level financial profiles, geographic
|
finance corpus. Provides unit-level financial profiles, geographic
|
||||||
rollups, and peer comparisons with auditable provenance and built-in
|
rollups, and peer comparisons with auditable provenance and built-in
|
||||||
cross-vintage correctness.
|
cross-vintage correctness.
|
||||||
License: MIT + file LICENSE
|
License: MIT + file LICENSE
|
||||||
|
URL: https://github.com/civilytics/uscogdata,
|
||||||
|
https://civilytics.r-universe.dev/uscogdata
|
||||||
|
BugReports: https://github.com/civilytics/uscogdata/issues
|
||||||
Encoding: UTF-8
|
Encoding: UTF-8
|
||||||
LazyData: false
|
LazyData: false
|
||||||
Depends: R (>= 4.1)
|
Depends: R (>= 4.1)
|
||||||
@@ -31,5 +38,5 @@ Suggests:
|
|||||||
Config/testthat/edition: 3
|
Config/testthat/edition: 3
|
||||||
VignetteBuilder: knitr
|
VignetteBuilder: knitr
|
||||||
RoxygenNote: 7.3.3
|
RoxygenNote: 7.3.3
|
||||||
MinCorpusSchema: 3
|
MinCorpusSchema: 4
|
||||||
MaxCorpusSchema: 3
|
MaxCorpusSchema: 7
|
||||||
|
|||||||
@@ -1,2 +1,2 @@
|
|||||||
YEAR: 2026
|
YEAR: 2026
|
||||||
COPYRIGHT HOLDER: Civilytics
|
COPYRIGHT HOLDER: Civilytics Consulting LLC
|
||||||
|
|||||||
+21
@@ -0,0 +1,21 @@
|
|||||||
|
# MIT License
|
||||||
|
|
||||||
|
Copyright (c) 2026 Civilytics Consulting LLC
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||||
|
of this software and associated documentation files (the "Software"), to deal
|
||||||
|
in the Software without restriction, including without limitation the rights
|
||||||
|
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||||
|
copies of the Software, and to permit persons to whom the Software is
|
||||||
|
furnished to do so, subject to the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be included in all
|
||||||
|
copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
|
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||||
|
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||||
|
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||||
|
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
SOFTWARE.
|
||||||
@@ -1,5 +1,6 @@
|
|||||||
# Generated by roxygen2: do not edit by hand
|
# Generated by roxygen2: do not edit by hand
|
||||||
|
|
||||||
|
export(cog_balances)
|
||||||
export(cog_basket_resolution)
|
export(cog_basket_resolution)
|
||||||
export(cog_basket_unresolved)
|
export(cog_basket_unresolved)
|
||||||
export(cog_categories)
|
export(cog_categories)
|
||||||
@@ -7,7 +8,9 @@ export(cog_explain)
|
|||||||
export(cog_find_peers)
|
export(cog_find_peers)
|
||||||
export(cog_geographic_rollup)
|
export(cog_geographic_rollup)
|
||||||
export(cog_gov_search)
|
export(cog_gov_search)
|
||||||
|
export(cog_manifest)
|
||||||
export(cog_mirror)
|
export(cog_mirror)
|
||||||
export(cog_peer_compare)
|
export(cog_peer_compare)
|
||||||
|
export(cog_recipes)
|
||||||
export(cog_revenue)
|
export(cog_revenue)
|
||||||
export(cog_spending)
|
export(cog_spending)
|
||||||
|
|||||||
@@ -1,21 +1,176 @@
|
|||||||
# uscogdata 0.1.0 (development)
|
# uscogdata 0.4.0
|
||||||
|
|
||||||
|
## DuckDB's resource budget is configurable
|
||||||
|
|
||||||
|
`USCOGDATA_DUCKDB_THREADS` and `USCOGDATA_DUCKDB_MEMORY_LIMIT` (with matching
|
||||||
|
`options(uscogdata.duckdb_threads = )` / `options(uscogdata.duckdb_memory_limit = )`
|
||||||
|
spellings) cap the DuckDB connection the package opens. Both follow the same
|
||||||
|
env-var > option > default precedence as `USCOGDATA_URL`.
|
||||||
|
|
||||||
|
Unset, **no pragma is issued at all** and DuckDB's own defaults apply exactly as
|
||||||
|
before -- every visible core. That is right for one interactive session on a
|
||||||
|
dedicated machine and wrong for a server: where several readers share a host, each
|
||||||
|
otherwise claims the whole machine and they contend. Capping measured ~5% on a
|
||||||
|
single-government all-years query (502 ms at 2 threads vs 475 ms uncapped on 16
|
||||||
|
cores), which is cheap enough that a server should always cap.
|
||||||
|
|
||||||
|
This replaces a workaround in which a consumer reached into the package namespace
|
||||||
|
at boot -- `getFromNamespace(".ensure_session", "uscogdata")()` followed by a manual
|
||||||
|
`SET threads` -- depending both on a private name and on the session already being
|
||||||
|
open.
|
||||||
|
|
||||||
|
## Cohorts can be named by predicate, not just by id
|
||||||
|
|
||||||
|
`cog_spending()`, `cog_revenue()` and `cog_balances()` gain optional `state`
|
||||||
|
and `type` arguments. Both default to `NULL`, so every existing call behaves
|
||||||
|
exactly as before.
|
||||||
|
|
||||||
|
Passing them expresses the cohort as a subquery against `canonical_fips_xwalk`
|
||||||
|
inside each statement, instead of round-tripping the ids through R and
|
||||||
|
rendering them back into a literal `IN` list:
|
||||||
|
|
||||||
|
```r
|
||||||
|
# before: resolve 20,106 ids in R, then embed them in every statement
|
||||||
|
ids <- cog_gov_search(NULL, state = "CA", type = "city")$canonical_govid
|
||||||
|
cog_spending(ids, years = 2022)
|
||||||
|
|
||||||
|
# now: the cohort never leaves the database
|
||||||
|
cog_spending(years = 2022, state = "CA", type = "city")
|
||||||
|
```
|
||||||
|
|
||||||
|
Measured against the production corpus, same FY2022 aggregate over the
|
||||||
|
20,106-government `type = "city"` cohort:
|
||||||
|
|
||||||
|
| cohort expressed as | time |
|
||||||
|
|---|---:|
|
||||||
|
| `IN (20,106 literals)` | 449 ms |
|
||||||
|
| join against a temp cohort table | 99 ms |
|
||||||
|
| predicate on `canonical_fips_xwalk` | **94 ms** |
|
||||||
|
| no cohort filter at all (the floor) | 88 ms |
|
||||||
|
|
||||||
|
**4.8x, within 7% of the floor.** The rendered `IN` list was 301,591
|
||||||
|
characters and was re-parsed in 5-8 separate statements per call, so the cost
|
||||||
|
was paid repeatedly; the predicate's size is constant in the cohort.
|
||||||
|
|
||||||
|
`state` and `type` use the same vocabulary and the same internal coercion as
|
||||||
|
`cog_gov_search()` -- `state` is a postal abbreviation (`"WI"`) even though the
|
||||||
|
crosswalk column holds a FIPS code (`"55"`).
|
||||||
|
|
||||||
|
Supplying `govid` **and** `state`/`type` intersects them: the governments in
|
||||||
|
`govid` that also match the predicate. Naming no cohort at all now aborts with
|
||||||
|
class `uscogdata_no_cohort` rather than R's "argument is missing" error.
|
||||||
|
|
||||||
|
When the cohort is named by predicate there is no id list to report, so
|
||||||
|
`provenance$scope$govids_found`/`govids_missing` are empty and
|
||||||
|
`provenance$scope$cohort` carries `state`, `type` and `n_governments` instead.
|
||||||
|
A `govid`-named cohort's provenance is unchanged.
|
||||||
|
|
||||||
|
## Fixes
|
||||||
|
|
||||||
|
* An unknown `state` abbreviation now aborts with "Unknown state abbreviation"
|
||||||
|
(class `uscogdata_unknown_state`) instead of base R's "subscript out of
|
||||||
|
bounds". `.state_abbrev_to_fips` is a named character vector, so `[[` on an
|
||||||
|
absent name threw before the curated message could be reached -- making that
|
||||||
|
message unreachable dead code in every verb that takes a `state`.
|
||||||
|
|
||||||
|
# uscogdata 0.3.0
|
||||||
|
|
||||||
|
First public release.
|
||||||
|
|
||||||
|
`uscogdata` provides curated R verbs over the Civilytics US Census of
|
||||||
|
Governments finance corpus: unit-level financial profiles, geographic rollups
|
||||||
|
and peer comparisons, with auditable provenance on every result.
|
||||||
|
|
||||||
|
## What it covers
|
||||||
|
|
||||||
|
Government types 0-3 (state, county, municipality, township), FY1967-FY2024 --
|
||||||
|
56 fiscal years, 46,148,034 rows, 190.6 MB. There is no source data for FY1968
|
||||||
|
or FY1969. Special districts (type 4) and school districts (type 5) are out of
|
||||||
|
scope pending validation.
|
||||||
|
|
||||||
|
## The verbs
|
||||||
|
|
||||||
|
`cog_spending()`, `cog_revenue()` and `cog_balances()` for flows and holdings;
|
||||||
|
`cog_gov_search()` to resolve place names (including basket mode for many at
|
||||||
|
once); `cog_find_peers()` and `cog_peer_compare()` for cohorts;
|
||||||
|
`cog_geographic_rollup()` for aggregates; `cog_categories()`, `cog_recipes()`,
|
||||||
|
`cog_manifest()` and `cog_explain()` for metadata and provenance; and
|
||||||
|
`cog_mirror()` for a local copy of the corpus.
|
||||||
|
|
||||||
|
## Reading the corpus now works out of the box
|
||||||
|
|
||||||
|
* The package reads the published corpus over HTTPS **with no configuration**.
|
||||||
|
Previously the default was a placeholder sentinel and no document in the
|
||||||
|
package supplied a working URL, so a new user had no path to a session.
|
||||||
|
* Remote reads work at all. The partitioned view used a glob, and DuckDB
|
||||||
|
cannot expand a glob over generic HTTP -- there is no directory listing to
|
||||||
|
expand against. Partition paths are now enumerated from the corpus manifest,
|
||||||
|
which is host-agnostic: an HTTPS mirror, a Nextcloud share and a local
|
||||||
|
`cog_mirror()` copy all take the same path.
|
||||||
|
* Nothing is written to disk in remote mode; DuckDB fetches only the row
|
||||||
|
groups a query needs.
|
||||||
|
|
||||||
|
## Four things to know before your first query
|
||||||
|
|
||||||
|
* **Amounts are in full US dollars.** The raw Census files report thousands;
|
||||||
|
the verbs multiply by 1000 on the way out. Do not multiply again.
|
||||||
|
* **Multi-government aggregates disclose their coverage.** The Census is a
|
||||||
|
complete enumeration only in years ending in 2 and 7; every other year is a
|
||||||
|
sample. Every such result carries `provenance$coverage` with per-year
|
||||||
|
`n_units_reporting`.
|
||||||
|
* **Absence means two different things.** Before FY2012 an absent cell means
|
||||||
|
Census published $0; from FY2012 it means not reported. `complete = TRUE`
|
||||||
|
labels which.
|
||||||
|
* **Series breaks reach you unasked.** Catalogued breaks intersecting your
|
||||||
|
query appear in provenance and in `cog_explain()`.
|
||||||
|
|
||||||
|
## Known limits
|
||||||
|
|
||||||
|
* Special districts (type 4) and school districts (type 5) are out of scope.
|
||||||
|
* Per-capita rollups exclude governments with no F-33 population, which is by
|
||||||
|
design but does silently narrow a rollup.
|
||||||
|
* `n_units_reporting` is category-conditional and is not a response rate.
|
||||||
|
* Employee-retirement (`X`) codes stop at FY2016, when those systems moved to
|
||||||
|
the Annual Survey of Public Pensions.
|
||||||
|
|
||||||
|
# uscogdata 0.2.0
|
||||||
|
|
||||||
## New features
|
## New features
|
||||||
|
|
||||||
* `cog_gov_search()` gains a **basket mode**: passing vector `name`
|
* `cog_spending()` and `cog_revenue()` accept the reserved category
|
||||||
/ `state` / `type` arguments resolves multiple place names in one
|
`"All Categories"`, returning one summed row per
|
||||||
call and returns a tibble of canonical rows in input order, ready
|
`(year, canonical_govid, subtype)` across every category inside the
|
||||||
to pipe into `cog_spending()` / `cog_revenue()`. Per-row resolution
|
requested concept's subtype scope. Filtering the result to
|
||||||
follows an exact-then-substring matching algorithm with deterministic
|
`spend_subtype == "operations"` gives an operating-expenditure total.
|
||||||
disambiguation; ambiguous and missing entries are surfaced via a
|
`cog_geographic_rollup()` inherits it,
|
||||||
sidecar audit tibble plus a single console summary message.
|
which is the efficient way to build a geographic total — previously a
|
||||||
* New exports `cog_basket_resolution()` and `cog_basket_unresolved()`
|
caller had to issue one rollup per category and sum the results
|
||||||
expose the basket sidecar for iterative query refinement.
|
(cog-api#37).
|
||||||
|
|
||||||
## Breaking changes
|
`"All Categories"` is not the same thing as `expenditure_concept = "total"`.
|
||||||
|
The concept chooses which subtypes are in scope; `"All Categories"` chooses
|
||||||
|
whether the rows inside that scope are broken out or summed.
|
||||||
|
|
||||||
* The first formal of `cog_gov_search()` was renamed from `pattern`
|
* `cog_categories()` advertises `"All Categories"` for the expenditure and
|
||||||
to `name`. All existing call sites in `cog_explorer/` and the
|
revenue vocabularies, so the reserved value is discoverable.
|
||||||
package itself use positional first-arg, so this rename is
|
|
||||||
non-breaking in practice. Callers that pass `pattern = ...` by name
|
* Coverage signposting (see "Signposting now catches partially-suppressed
|
||||||
must update to `name = ...`.
|
categories" below) now also works in `category = "All Categories"` mode.
|
||||||
|
The recipe-suggestion candidate query used to be scoped by `category`,
|
||||||
|
which is never a match for the reserved `"All Categories"` value, so
|
||||||
|
`provenance$suggestions` always came back empty there — the one mode whose
|
||||||
|
whole point is "you cannot sum the wrong scope" was silently unable to
|
||||||
|
signal a wrong scope. The candidate query is now scoped by the concept's
|
||||||
|
subtype allowlist instead, symmetric with how `.build_verb_sql()` itself
|
||||||
|
scopes the summed total: Los Angeles County FY2011, `category = "All
|
||||||
|
Categories"` still excludes $271,589,000 of aggregate-published Public
|
||||||
|
Welfare (`E68`), but now names `recipe = "welfare_cash_e68_wide"` to
|
||||||
|
recover it instead of reporting zero suggestions.
|
||||||
|
|
||||||
|
## Documentation
|
||||||
|
|
||||||
|
* `cog_geographic_rollup()` and `cog_peer_compare()` now document that
|
||||||
|
`provenance$coverage`'s `n_units_reporting` is **category-conditional** and
|
||||||
|
is not a response rate: a government that was surveyed and genuinely spends
|
||||||
|
nothing in the requested category is indistinguishable from one never
|
||||||
|
surveyed (uscogdata#36).
|
||||||
|
|||||||
@@ -0,0 +1,122 @@
|
|||||||
|
# R/balance_caveats.R
|
||||||
|
#
|
||||||
|
# The four caveats from cog_pipeline/docs/data_dictionary.md § Cash and
|
||||||
|
# security holdings. Each one silently invalidates an obvious analysis, so
|
||||||
|
# they travel in provenance (machine-readable, for cog-api#26) rather than
|
||||||
|
# living only in prose.
|
||||||
|
#
|
||||||
|
# Two of the four are already carried by the code-driven series-break
|
||||||
|
# builders and are deliberately NOT duplicated here:
|
||||||
|
# * SB195/SB196 -- X40/X41 book -> market at FY2002 -- fire via
|
||||||
|
# series_break_refs on the recipe path, the only path that observes those
|
||||||
|
# codes.
|
||||||
|
# What remains is the GAAP distinction (a constant) and the coverage windows
|
||||||
|
# (measured, never hardcoded, so they stay correct as the corpus grows).
|
||||||
|
|
||||||
|
#' Per-subtype observed year extents, plus which requested families are
|
||||||
|
#' truncated relative to the requested span.
|
||||||
|
#' @noRd
|
||||||
|
.balance_caveats <- function(con, codes_observed, years) {
|
||||||
|
cw <- .balance_coverage_windows(con)
|
||||||
|
|
||||||
|
observed_subtypes <- if (length(codes_observed) == 0L) {
|
||||||
|
character(0)
|
||||||
|
} else {
|
||||||
|
DBI::dbGetQuery(con, sprintf(
|
||||||
|
"SELECT DISTINCT balance_subtype FROM summary_categories
|
||||||
|
WHERE item_code IN (%s) AND balance_subtype IS NOT NULL",
|
||||||
|
.sql_lit_chr(codes_observed)
|
||||||
|
))$balance_subtype
|
||||||
|
}
|
||||||
|
|
||||||
|
# A family is "truncated" when the caller asked for years outside the span
|
||||||
|
# that family actually covers -- the FY2016 employee-retirement termination
|
||||||
|
# and the FY2021 end of the W family are both this shape.
|
||||||
|
truncated <- character(0)
|
||||||
|
if (length(years) > 0L) {
|
||||||
|
for (s in observed_subtypes) {
|
||||||
|
w <- cw[[s]]
|
||||||
|
if (is.null(w)) next
|
||||||
|
if (max(years) > w[2] || min(years) < w[1]) truncated <- c(truncated, s)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
list(
|
||||||
|
not_gaap = TRUE,
|
||||||
|
not_gaap_note = paste0(
|
||||||
|
"Census holdings are gross -- no liabilities are netted -- and are NOT ",
|
||||||
|
"GAAP fund balance. A reserve ratio built from them overstates what is ",
|
||||||
|
"actually available."
|
||||||
|
),
|
||||||
|
coverage_window = cw,
|
||||||
|
truncated = sort(unique(truncated))
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Per-subtype [min year, max year] extents for EVERY balance subtype in the
|
||||||
|
#' mounted corpus, memoised for the session.
|
||||||
|
#'
|
||||||
|
#' The query carries no govid and no year predicate -- its answer is a property
|
||||||
|
#' of the mounted corpus alone and cannot change between calls -- but it scans
|
||||||
|
#' the whole of `balance_long`, which measured 35% of `cog_balances()` runtime
|
||||||
|
#' on the bundled fixture and would be a per-request throughput ceiling once
|
||||||
|
#' cog-api#26 serves this verb over HTTP. Memoised in `.uscogdata_env` and
|
||||||
|
#' invalidated by `cog_close()`, the same pattern as `.uscogdata_env$manifest`.
|
||||||
|
#'
|
||||||
|
#' Scope is deliberately corpus-wide rather than query-scoped: a caller asking
|
||||||
|
#' "is there a family I missed?" needs every window. The observed-scoped field
|
||||||
|
#' is `truncated`. Documented as such in inst/schemas/provenance-v1.json.
|
||||||
|
#' @noRd
|
||||||
|
.balance_coverage_windows <- function(con) {
|
||||||
|
cached <- .uscogdata_env$balance_coverage_windows
|
||||||
|
if (!is.null(cached)) return(cached)
|
||||||
|
|
||||||
|
windows <- DBI::dbGetQuery(con,
|
||||||
|
"SELECT c.balance_subtype AS subtype,
|
||||||
|
MIN(l.year) AS year_min,
|
||||||
|
MAX(l.year) AS year_max
|
||||||
|
FROM balance_long l
|
||||||
|
JOIN summary_categories c USING (item_code)
|
||||||
|
WHERE c.balance_subtype IS NOT NULL
|
||||||
|
GROUP BY 1
|
||||||
|
ORDER BY 1"
|
||||||
|
)
|
||||||
|
|
||||||
|
cw <- stats::setNames(
|
||||||
|
lapply(seq_len(nrow(windows)),
|
||||||
|
function(i) as.integer(c(windows$year_min[i], windows$year_max[i]))),
|
||||||
|
windows$subtype
|
||||||
|
)
|
||||||
|
.uscogdata_env$balance_coverage_windows <- cw
|
||||||
|
cw
|
||||||
|
}
|
||||||
|
|
||||||
|
#' TRUE the first time `key` is seen this session, FALSE thereafter.
|
||||||
|
#' Reset by cog_close().
|
||||||
|
#' @noRd
|
||||||
|
.balance_caveat_once <- function(key) {
|
||||||
|
seen <- .uscogdata_env$balance_caveats_shown
|
||||||
|
if (is.null(seen)) seen <- character(0)
|
||||||
|
if (key %in% seen) return(FALSE)
|
||||||
|
.uscogdata_env$balance_caveats_shown <- c(seen, key)
|
||||||
|
TRUE
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Emit at most one message per caveat class per session.
|
||||||
|
#' @noRd
|
||||||
|
.emit_balance_caveats <- function(caveats) {
|
||||||
|
if (.balance_caveat_once("not_gaap")) {
|
||||||
|
cli::cli_inform(c(
|
||||||
|
"!" = "Census holdings are gross and are {.strong not} GAAP fund balance.",
|
||||||
|
"i" = "No liabilities are netted; a reserve ratio built from them overstates available funds."
|
||||||
|
))
|
||||||
|
}
|
||||||
|
if (length(caveats$truncated) > 0L &&
|
||||||
|
.balance_caveat_once("coverage_window")) {
|
||||||
|
cli::cli_inform(c(
|
||||||
|
"!" = "Requested years extend beyond what {.val {caveats$truncated}} actually covers.",
|
||||||
|
"i" = "See {.code provenance$balance_caveats$coverage_window}."
|
||||||
|
))
|
||||||
|
}
|
||||||
|
invisible(NULL)
|
||||||
|
}
|
||||||
+177
@@ -0,0 +1,177 @@
|
|||||||
|
# R/balances.R
|
||||||
|
#
|
||||||
|
# Cash and security holdings. A third verb rather than an argument on a money
|
||||||
|
# verb because holdings are a STOCK -- a balance at a point in time -- while
|
||||||
|
# cog_spending()/cog_revenue() return FLOWS over a fiscal year. The money
|
||||||
|
# verbs' whole argument vocabulary (expenditure_concept, revenue_concept,
|
||||||
|
# complete=) describes flows and is meaningless here, so this deliberately
|
||||||
|
# does NOT route through .verb_spendrev().
|
||||||
|
|
||||||
|
#' Cash and security holdings for one or more governments
|
||||||
|
#'
|
||||||
|
#' Returns Census cash-and-security holdings (`category_type = "balance"`):
|
||||||
|
#' fund balances, retirement system holdings and insurance trust balances.
|
||||||
|
#'
|
||||||
|
#' @section Holdings are not GAAP fund balance:
|
||||||
|
#' Census holdings are **gross** -- no liabilities are netted -- so a reserve
|
||||||
|
#' ratio built from them overstates what is actually available. They are not
|
||||||
|
#' comparable to a GAAP fund balance from an ACFR.
|
||||||
|
#'
|
||||||
|
#' @param govid Canonical govid(s): a character vector, or a data frame with a
|
||||||
|
#' `canonical_govid` column (e.g. from [cog_gov_search()]). `NULL` to name
|
||||||
|
#' the cohort by `state`/`type` instead.
|
||||||
|
#' @inheritParams cog_spending
|
||||||
|
#' @param years Integer vector of fiscal years.
|
||||||
|
#' @param category Optional character vector of categories to keep. One of
|
||||||
|
#' `"Fund Balances"`, `"Insurance Trust Balances"`,
|
||||||
|
#' `"Retirement System Holdings"`. There is deliberately no `subtype`
|
||||||
|
#' argument: for holdings, `category` is a strict coarsening of
|
||||||
|
#' `balance_subtype` (unlike the money verbs, where the two axes cross), so
|
||||||
|
#' every combination would be either redundant or empty.
|
||||||
|
#' `category = "Fund Balances"` is exactly the `general` family
|
||||||
|
#' (`W01`/`W31`/`W61`). `balance_subtype` is returned, so a finer split is
|
||||||
|
#' one `dplyr::filter()` away. The reserved pseudo-category
|
||||||
|
#' `"All Categories"` (see [cog_spending()]) is **not** supported here and
|
||||||
|
#' errors with class `uscogdata_all_categories_unsupported`: it sums a
|
||||||
|
#' concept's subtype scope, and holdings are a stock with no concept
|
||||||
|
#' vocabulary to sum across. Omit `category` to get every category broken
|
||||||
|
#' out instead.
|
||||||
|
#' @param per_capita Divide holdings by population. Note this is a **stock per
|
||||||
|
#' resident** (reserves per person), which is *not* comparable to
|
||||||
|
#' [cog_spending()]'s per-capita figures -- those are a flow per person.
|
||||||
|
#' @param adjust_to_year Deflate to this year's dollars (CPI-U).
|
||||||
|
#' @param basis Accepted for uniformity with the money verbs, but currently a
|
||||||
|
#' **no-op**: `harmonization_map` carries no balance-code rows, so harmonized
|
||||||
|
#' and raw space are identical for holdings. Reported in
|
||||||
|
#' `provenance$basis_note`.
|
||||||
|
#' @param recipe Optional harmonization recipe id (see [cog_recipes()]).
|
||||||
|
#' `"cash_securities_z77_wide"` and `"cash_securities_z78_wide"` bridge the
|
||||||
|
#' wide era to the modern one.
|
||||||
|
#'
|
||||||
|
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||||
|
#' `balance_subtype`, `category`, `amt_nominal`, `codes_included`,
|
||||||
|
#' `aggregate_fallback`, plus optional `amt_per_capita_nominal` and
|
||||||
|
#' `pop_source` (when `per_capita = TRUE`), optional `amt_real` (when
|
||||||
|
#' `adjust_to_year` is set), and optional `amt_per_capita_real` (only when
|
||||||
|
#' **both** `per_capita = TRUE` and `adjust_to_year` are set -- there is no
|
||||||
|
#' nominal per-capita column to deflate otherwise). Amounts are full US
|
||||||
|
#' dollars.
|
||||||
|
#'
|
||||||
|
#' Carries a `provenance` attribute matching
|
||||||
|
#' `inst/schemas/provenance-v1.json`, whose `balance_caveats` block reports
|
||||||
|
#' `not_gaap`, `not_gaap_note`, `coverage_window` (measured year extents for
|
||||||
|
#' every balance subtype in the mounted corpus, not only the observed ones)
|
||||||
|
#' and `truncated` (the observed subtypes whose coverage falls short of the
|
||||||
|
#' requested years). `expenditure_concept`/`revenue_concept` are `NA` --
|
||||||
|
#' holdings are a stock, not a flow, so neither concept vocabulary applies.
|
||||||
|
#' @export
|
||||||
|
cog_balances <- function(govid = NULL, years, category = NULL,
|
||||||
|
per_capita = FALSE, adjust_to_year = NULL,
|
||||||
|
basis = c("harmonized", "raw"), recipe = NULL,
|
||||||
|
state = NULL, type = NULL) {
|
||||||
|
call <- match.call()
|
||||||
|
basis <- match.arg(basis, c("harmonized", "raw"))
|
||||||
|
# Coerce FIRST, validate second: .validate_verb_inputs() asserts
|
||||||
|
# is.character(govid), and a data-frame govid (cog_gov_search() output) has
|
||||||
|
# not been unwrapped yet at this point.
|
||||||
|
govid <- if (is.null(govid)) NULL else .coerce_govid_input(govid)
|
||||||
|
# The money verbs' validator, reused rather than re-implemented (R/spending.R).
|
||||||
|
# It covers the exact superset cog_balances() needs -- including the
|
||||||
|
# recipe/category mutual-exclusivity guard -- so a second local copy would
|
||||||
|
# only be a place for the two to drift apart. This is the same kind of
|
||||||
|
# helper reuse as .build_verb_sql()/.attach_per_capita() below; it does NOT
|
||||||
|
# route the verb through .verb_spendrev(), which stays deliberately unused
|
||||||
|
# here because its flow vocabulary is meaningless for a stock.
|
||||||
|
#
|
||||||
|
# allow_all_categories is left at its FALSE default (contrast
|
||||||
|
# .verb_spendrev(), which passes TRUE): the all-categories mode's "sum"
|
||||||
|
# only means something in terms of a concept's subtype scope, and holdings
|
||||||
|
# have no concept vocabulary. The reuse above is exactly why this can be a
|
||||||
|
# one-line default rather than a second bespoke check -- see the
|
||||||
|
# validator's own doc comment for the incident that made that matter.
|
||||||
|
.validate_verb_inputs(govid, years, category, per_capita, adjust_to_year,
|
||||||
|
recipe)
|
||||||
|
years <- as.integer(years)
|
||||||
|
if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year)
|
||||||
|
|
||||||
|
cohort <- .make_cohort(govid, state, type)
|
||||||
|
|
||||||
|
con <- .ensure_session()
|
||||||
|
.require_balance_support(con)
|
||||||
|
scope <- .check_govids_in_scope(govid)
|
||||||
|
|
||||||
|
basis_note <- paste0(
|
||||||
|
"`basis` has no effect on holdings: harmonization_map carries no ",
|
||||||
|
"balance-code rows, so harmonized and raw space are identical here."
|
||||||
|
)
|
||||||
|
|
||||||
|
manifest <- .uscogdata_env$manifest
|
||||||
|
recipe_block <- NULL
|
||||||
|
category_for_prov <- category
|
||||||
|
|
||||||
|
if (!is.null(recipe)) {
|
||||||
|
.require_schema_v5(con, manifest, "recipe =")
|
||||||
|
.validate_recipe_id(con, recipe)
|
||||||
|
comps <- .recipe_components(con, recipe)
|
||||||
|
recipe_label <- comps$label[[1]]
|
||||||
|
result <- .run_recipe(con, recipe, cohort, years)
|
||||||
|
sql <- attr(result, "sql_query")
|
||||||
|
result <- .shape_recipe_result(result, "balance_subtype", recipe_label)
|
||||||
|
recipe_block <- list(
|
||||||
|
recipe_id = recipe, label = recipe_label,
|
||||||
|
components = .df_to_row_list(comps)
|
||||||
|
)
|
||||||
|
category_for_prov <- recipe_label
|
||||||
|
} else {
|
||||||
|
sql <- .build_verb_sql("balance_annotated", "balance_subtype",
|
||||||
|
cohort, years, category,
|
||||||
|
ig_view = NULL, subtype_scope = NULL)
|
||||||
|
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||||
|
}
|
||||||
|
|
||||||
|
# Order matters (matches .verb_spendrev()): per-capita first, so
|
||||||
|
# .attach_real_dollars() deflates the nominal per-capita column into
|
||||||
|
# amt_per_capita_real rather than needing amt_per_capita_nominal recomputed.
|
||||||
|
if (isTRUE(per_capita)) result <- .attach_per_capita(result, con)
|
||||||
|
if (!is.null(adjust_to_year)) {
|
||||||
|
result <- .attach_real_dollars(result, adjust_to_year, per_capita)
|
||||||
|
}
|
||||||
|
|
||||||
|
prov <- .build_provenance(
|
||||||
|
verb = "cog_balances", call = call, govid = govid, years = years,
|
||||||
|
category = category_for_prov, per_capita = per_capita,
|
||||||
|
adjust_to_year = adjust_to_year, result = result, sql = sql,
|
||||||
|
subtype_col = "balance_subtype",
|
||||||
|
basis = basis, basis_note = basis_note,
|
||||||
|
# Neither concept vocabulary applies to a stock.
|
||||||
|
expenditure_concept = NA_character_,
|
||||||
|
revenue_concept = NA_character_,
|
||||||
|
recipe = recipe_block
|
||||||
|
)
|
||||||
|
prov$scope$govids_found <- scope$found
|
||||||
|
prov$scope$govids_missing <- scope$missing
|
||||||
|
prov$scope$cohort <- .cohort_provenance(con, cohort)
|
||||||
|
|
||||||
|
prov$balance_caveats <- .balance_caveats(
|
||||||
|
con, prov$codes_summed$observed, years
|
||||||
|
)
|
||||||
|
.emit_balance_caveats(prov$balance_caveats)
|
||||||
|
|
||||||
|
attr(result, "provenance") <- prov
|
||||||
|
result
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Abort unless the mounted corpus classifies balance codes.
|
||||||
|
#'
|
||||||
|
#' `balance_subtype` arrived with cog_pipeline #76/#77 without a
|
||||||
|
#' schema_version bump, so the check is on the column, not the version.
|
||||||
|
#' @noRd
|
||||||
|
.require_balance_support <- function(con) {
|
||||||
|
if (.corpus_has_balance_subtype(con)) return(invisible(TRUE))
|
||||||
|
cli::cli_abort(
|
||||||
|
c("This corpus does not classify cash and security holdings.",
|
||||||
|
i = "`summary_categories` has no {.field balance_subtype} column.",
|
||||||
|
i = "Republish from cog_pipeline at #76/#77 or later."),
|
||||||
|
class = "uscogdata_no_balance_support"
|
||||||
|
)
|
||||||
|
}
|
||||||
@@ -0,0 +1,87 @@
|
|||||||
|
# R/basis.R
|
||||||
|
# basis= resolution (harmonized/raw, with v4/v5 dual-accept) and the
|
||||||
|
# harmonization exclusion-count block attached to provenance.
|
||||||
|
|
||||||
|
#' Resolve the requested `basis` against the active corpus's schema_version.
|
||||||
|
#'
|
||||||
|
#' On a `schema_version >= 5` corpus, the requested basis is used as-is. On
|
||||||
|
#' an older (`schema_version == 4`) corpus, which has no harmonization
|
||||||
|
#' tables: a caller who left `basis` at its default (`"harmonized"`, so
|
||||||
|
#' `explicit` is `FALSE`) silently gets `"raw"` back, with a note recorded
|
||||||
|
#' for provenance; a caller who explicitly asked for
|
||||||
|
#' `basis = "harmonized"` gets a hard abort instead of a silent downgrade.
|
||||||
|
#'
|
||||||
|
#' @param basis `"harmonized"` or `"raw"` (already resolved via `match.arg`).
|
||||||
|
#' @param explicit `TRUE` if the caller passed `basis` explicitly (as
|
||||||
|
#' opposed to relying on the default `c("harmonized", "raw")`).
|
||||||
|
#' @param manifest The active session's parsed manifest list.
|
||||||
|
#' @return List with `basis` (the resolved value) and `note` (character or
|
||||||
|
#' `NA_character_`).
|
||||||
|
#' @noRd
|
||||||
|
.resolve_basis <- function(basis, explicit, manifest) {
|
||||||
|
schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L))
|
||||||
|
|
||||||
|
if (schema_version >= 5L) {
|
||||||
|
return(list(basis = basis, note = NA_character_))
|
||||||
|
}
|
||||||
|
|
||||||
|
if (identical(basis, "harmonized") && explicit) {
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"basis = \"harmonized\" requires corpus schema_version >= 5.",
|
||||||
|
x = "Active corpus has schema_version {schema_version}.",
|
||||||
|
i = "Use basis = \"raw\" (the default on this corpus), or point USCOGDATA_URL at a schema_version >= 5 corpus."
|
||||||
|
), class = "uscogdata_basis_unsupported")
|
||||||
|
}
|
||||||
|
|
||||||
|
list(
|
||||||
|
basis = "raw",
|
||||||
|
note = sprintf(
|
||||||
|
"basis resolved to \"raw\": corpus schema_version %d < 5 (harmonization tables unavailable)",
|
||||||
|
schema_version
|
||||||
|
)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Count + sum item-level rows that basis="harmonized" excludes because they
|
||||||
|
#' carry no harmonized_code (discontinued / not-yet-ruled codes) within the
|
||||||
|
#' calling verb's crosswalk scope (`subtype_col` values in `subtype_scope` --
|
||||||
|
#' the same subtype-membership classification the verb SQL uses, never
|
||||||
|
#' item-code prefixes), govids, and years. Only meaningful when the resolved
|
||||||
|
#' basis is "harmonized"; returns an applied = FALSE stub otherwise (raw
|
||||||
|
#' basis never excludes rows this way).
|
||||||
|
#'
|
||||||
|
#' The intergovernmental leg is deliberately outside this count even for
|
||||||
|
#' expenditure_concept = "total": ig_long_harmonized COALESCEs rather than
|
||||||
|
#' drops NULL-harmonized rows, so harmonization never excludes an IG row.
|
||||||
|
#' @noRd
|
||||||
|
.build_harmonization_block <- function(con, cohort, years, resolved,
|
||||||
|
subtype_col, subtype_scope) {
|
||||||
|
if (!identical(resolved$basis, "harmonized")) {
|
||||||
|
return(list(
|
||||||
|
applied = FALSE,
|
||||||
|
na_rows_excluded = 0L,
|
||||||
|
na_amount_excluded = 0,
|
||||||
|
note = resolved$note
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
|
sql <- sprintf(
|
||||||
|
"SELECT COUNT(*) AS n, COALESCE(SUM(amt), 0) * 1000.0 AS amt
|
||||||
|
FROM long
|
||||||
|
WHERE %s AND year IN (%s)
|
||||||
|
AND NOT is_aggregate AND harmonized_code IS NULL
|
||||||
|
AND item_code IN (
|
||||||
|
SELECT item_code FROM summary_categories WHERE %s IN (%s)
|
||||||
|
)",
|
||||||
|
.cohort_sql(cohort), paste(as.integer(years), collapse = ","),
|
||||||
|
subtype_col, .sql_lit_chr(subtype_scope)
|
||||||
|
)
|
||||||
|
na <- DBI::dbGetQuery(con, sql)
|
||||||
|
|
||||||
|
list(
|
||||||
|
applied = TRUE,
|
||||||
|
na_rows_excluded = as.integer(na$n),
|
||||||
|
na_amount_excluded = as.numeric(na$amt),
|
||||||
|
note = resolved$note
|
||||||
|
)
|
||||||
|
}
|
||||||
+41
-8
@@ -5,23 +5,33 @@
|
|||||||
#' Returns the category taxonomy exposed by the corpus's
|
#' Returns the category taxonomy exposed by the corpus's
|
||||||
#' `summary_categories` view, grouped to one row per
|
#' `summary_categories` view, grouped to one row per
|
||||||
#' `(category, subtype)` pair. Use this to discover valid `category`
|
#' `(category, subtype)` pair. Use this to discover valid `category`
|
||||||
#' values for [cog_spending()] / [cog_revenue()] /
|
#' values for [cog_spending()] / [cog_revenue()] / [cog_balances()] /
|
||||||
#' [cog_geographic_rollup()] and to audit which Census item codes feed
|
#' [cog_geographic_rollup()] and to audit which Census item codes feed
|
||||||
#' each category.
|
#' each category.
|
||||||
#'
|
#'
|
||||||
#' @param type Either `NULL` (default, return both spending and revenue
|
#' `subtype` COALESCEs the crosswalk's three subtype columns, so it carries
|
||||||
#' rows), `"spending"`, or `"revenue"`.
|
#' `spend_subtype` on expenditure rows, `revenue_subtype` on revenue rows and
|
||||||
|
#' `balance_subtype` on balance rows. Note that [cog_balances()] itself takes
|
||||||
|
#' no `subtype` argument — for holdings, `category` is a strict coarsening of
|
||||||
|
#' `balance_subtype` — but the value is surfaced here because it is the
|
||||||
|
#' discovery surface downstream consumers build their vocabulary from.
|
||||||
|
#'
|
||||||
|
#' @param type Either `NULL` (default, every row: expenditure, revenue and
|
||||||
|
#' balance), `"spending"`, `"revenue"`, or `"balance"`.
|
||||||
#' @param pattern Optional regex matched case-insensitively against the
|
#' @param pattern Optional regex matched case-insensitively against the
|
||||||
#' `category` column (e.g. `"Police"` or `"Tax"`).
|
#' `category` column (e.g. `"Police"` or `"Tax"`).
|
||||||
#' @return Tibble with columns `category`, `category_type`, `subtype`,
|
#' @return Tibble with columns `category`, `category_type`, `subtype`,
|
||||||
#' `n_codes`, `item_codes` (comma-separated, alphabetical). Sorted by
|
#' `n_codes`, `item_codes` (comma-separated, alphabetical). Sorted by
|
||||||
#' `category_type`, `category`, `subtype`.
|
#' `category_type`, `category`, `subtype`. Includes one row per flow for the
|
||||||
|
#' reserved pseudo-category `"All Categories"`, which carries `NA` for
|
||||||
|
#' `subtype`, `n_codes` and `item_codes` because it is a query mode rather
|
||||||
|
#' than a crosswalk entry — see [cog_spending()]'s `category` argument.
|
||||||
#' @export
|
#' @export
|
||||||
cog_categories <- function(type = NULL, pattern = NULL) {
|
cog_categories <- function(type = NULL, pattern = NULL) {
|
||||||
if (!is.null(type)) {
|
if (!is.null(type)) {
|
||||||
if (!is.character(type) || length(type) != 1L ||
|
if (!is.character(type) || length(type) != 1L ||
|
||||||
!type %in% c("spending", "revenue")) {
|
!type %in% c("spending", "revenue", "balance")) {
|
||||||
cli::cli_abort('`type` must be NULL, "spending", or "revenue".')
|
cli::cli_abort('`type` must be NULL, "spending", "revenue", or "balance".')
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (!is.null(pattern) &&
|
if (!is.null(pattern) &&
|
||||||
@@ -48,7 +58,7 @@ cog_categories <- function(type = NULL, pattern = NULL) {
|
|||||||
|
|
||||||
sql <- paste(
|
sql <- paste(
|
||||||
"SELECT category, category_type,
|
"SELECT category, category_type,
|
||||||
COALESCE(spend_subtype, revenue_subtype) AS subtype,
|
COALESCE(spend_subtype, revenue_subtype, balance_subtype) AS subtype,
|
||||||
COUNT(DISTINCT item_code) AS n_codes,
|
COUNT(DISTINCT item_code) AS n_codes,
|
||||||
string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS item_codes
|
string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS item_codes
|
||||||
FROM summary_categories",
|
FROM summary_categories",
|
||||||
@@ -56,5 +66,28 @@ cog_categories <- function(type = NULL, pattern = NULL) {
|
|||||||
"GROUP BY category, category_type, subtype
|
"GROUP BY category, category_type, subtype
|
||||||
ORDER BY category_type, category, subtype"
|
ORDER BY category_type, category, subtype"
|
||||||
)
|
)
|
||||||
tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
out <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||||
|
|
||||||
|
# The reserved pseudo-category is a query mode, not a crosswalk row, so it
|
||||||
|
# has no item codes to report -- hence NA rather than 0 for n_codes. It is
|
||||||
|
# emitted for the two FLOW vocabularies only: cog_balances() returns a stock
|
||||||
|
# and has no concept argument to sum within.
|
||||||
|
pseudo <- tibble::tibble(
|
||||||
|
category = .ALL_CATEGORIES,
|
||||||
|
category_type = c("expenditure", "revenue"),
|
||||||
|
subtype = NA_character_,
|
||||||
|
n_codes = NA_integer_,
|
||||||
|
item_codes = NA_character_
|
||||||
|
)
|
||||||
|
if (!is.null(type)) {
|
||||||
|
db_type <- if (type == "spending") "expenditure" else type
|
||||||
|
pseudo <- pseudo[pseudo$category_type == db_type, , drop = FALSE]
|
||||||
|
}
|
||||||
|
if (!is.null(pattern) && nrow(pseudo) > 0L) {
|
||||||
|
keep <- grepl(pattern, pseudo$category, ignore.case = TRUE)
|
||||||
|
pseudo <- pseudo[keep, , drop = FALSE]
|
||||||
|
}
|
||||||
|
if (nrow(pseudo) == 0L) return(out)
|
||||||
|
out <- rbind(out, pseudo)
|
||||||
|
out[order(out$category_type, out$category, out$subtype), , drop = FALSE]
|
||||||
}
|
}
|
||||||
|
|||||||
+127
@@ -0,0 +1,127 @@
|
|||||||
|
# How a verb names the set of governments it queries.
|
||||||
|
#
|
||||||
|
# Historically there was one way: a `govid` character vector, rendered by
|
||||||
|
# .sql_lit_chr() into a quoted IN list. That is fine for a handful of
|
||||||
|
# governments and pathological for a fleet. Measured against the production
|
||||||
|
# corpus, the same FY2022 aggregate over the 20,106-government `type = "city"`
|
||||||
|
# cohort:
|
||||||
|
#
|
||||||
|
# cohort expressed as time
|
||||||
|
# IN (20,106 literals) 449 ms
|
||||||
|
# join against a temp cohort table 99 ms
|
||||||
|
# predicate on canonical_fips_xwalk 94 ms
|
||||||
|
# no cohort filter at all (the floor) 88 ms
|
||||||
|
#
|
||||||
|
# 4.8x, and within 7% of the no-filter floor. The rendered IN list is 301,591
|
||||||
|
# characters and .verb_spendrev() embeds it in 5-8 separate statements per
|
||||||
|
# call, so the parse-and-plan cost is paid over and over (uscogdata#58).
|
||||||
|
#
|
||||||
|
# A cohort therefore has two independent halves, and a query can carry either
|
||||||
|
# or both:
|
||||||
|
#
|
||||||
|
# ids an explicit canonical_govid vector -> literal IN list
|
||||||
|
# predicate state/type over canonical_fips_xwalk -> IN (SELECT ...)
|
||||||
|
#
|
||||||
|
# Both together is an INTERSECTION -- "these ids, narrowed to that state/type"
|
||||||
|
# -- never a precedence rule where one silently wins.
|
||||||
|
|
||||||
|
#' Build the internal cohort object shared by every query verb.
|
||||||
|
#'
|
||||||
|
#' `state` and `type` are coerced with the SAME helpers `cog_gov_search()`
|
||||||
|
#' uses. That is load-bearing, not tidiness: the public argument is a postal
|
||||||
|
#' abbreviation (`"WI"`) while `canonical_fips_xwalk.fips_state` holds a FIPS
|
||||||
|
#' code (`"55"`), and `type` is a label (`"city"`) against an integer
|
||||||
|
#' `govs_type`. A predicate written against the raw parameter matches nothing
|
||||||
|
#' and returns an empty result indistinguishable from "this government
|
||||||
|
#' reported nothing" -- cog-api hit exactly that trap optimizing this path.
|
||||||
|
#' One definition of the translation, not two.
|
||||||
|
#'
|
||||||
|
#' @param govid Already-coerced character vector of canonical_govids, or NULL.
|
||||||
|
#' @param state Postal abbreviation or FIPS code, or NULL.
|
||||||
|
#' @param type Type label or integer code, or NULL.
|
||||||
|
#' @noRd
|
||||||
|
.make_cohort <- function(govid = NULL, state = NULL, type = NULL) {
|
||||||
|
if (is.null(govid) && is.null(state) && is.null(type)) {
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"A cohort must be named.",
|
||||||
|
"*" = "Pass {.arg govid} for specific governments, or {.arg state}/{.arg type} for every government matching a predicate.",
|
||||||
|
"i" = "Passing both intersects them: the governments in {.arg govid} that also match {.arg state}/{.arg type}."
|
||||||
|
), class = "uscogdata_no_cohort")
|
||||||
|
}
|
||||||
|
structure(
|
||||||
|
list(
|
||||||
|
ids = govid,
|
||||||
|
state = state,
|
||||||
|
type = type,
|
||||||
|
state_fips = if (is.null(state)) NULL else .coerce_state_to_fips(state),
|
||||||
|
type_int = if (is.null(type)) NULL else .coerce_type(type)
|
||||||
|
),
|
||||||
|
class = "uscogdata_cohort"
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Is any part of this cohort expressed as an xwalk predicate?
|
||||||
|
#' @noRd
|
||||||
|
.cohort_by_predicate <- function(cohort) {
|
||||||
|
!is.null(cohort$state_fips) || !is.null(cohort$type_int)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Render the cohort as a SQL boolean expression over `col`.
|
||||||
|
#'
|
||||||
|
#' `col` may be qualified (`"l.canonical_govid"`, `"x.canonical_govid"`) --
|
||||||
|
#' several call sites join the xwalk under an alias. The subquery's own
|
||||||
|
#' projected column stays unqualified: it selects from canonical_fips_xwalk,
|
||||||
|
#' not from the outer relation.
|
||||||
|
#' @noRd
|
||||||
|
.cohort_sql <- function(cohort, col = "canonical_govid") {
|
||||||
|
preds <- character(0)
|
||||||
|
|
||||||
|
if (!is.null(cohort$ids)) {
|
||||||
|
preds <- c(preds, sprintf("%s IN (%s)", col, .sql_lit_chr(cohort$ids)))
|
||||||
|
}
|
||||||
|
|
||||||
|
if (.cohort_by_predicate(cohort)) {
|
||||||
|
xwalk_preds <- character(0)
|
||||||
|
if (!is.null(cohort$state_fips)) {
|
||||||
|
xwalk_preds <- c(xwalk_preds,
|
||||||
|
sprintf("fips_state = %s", .sql_lit_chr(cohort$state_fips)))
|
||||||
|
}
|
||||||
|
if (!is.null(cohort$type_int)) {
|
||||||
|
xwalk_preds <- c(xwalk_preds, sprintf("govs_type = %d", cohort$type_int))
|
||||||
|
}
|
||||||
|
preds <- c(preds, sprintf(
|
||||||
|
"%s IN (SELECT canonical_govid FROM canonical_fips_xwalk WHERE %s)",
|
||||||
|
col, paste(xwalk_preds, collapse = " AND ")
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
|
paste(preds, collapse = " AND ")
|
||||||
|
}
|
||||||
|
|
||||||
|
#' How many governments the cohort covers.
|
||||||
|
#'
|
||||||
|
#' One COUNT against the crosswalk, used only to populate the provenance
|
||||||
|
#' `scope$cohort` block. Deliberately a count rather than the id list: a
|
||||||
|
#' fleet-scale cohort would otherwise put 20,000 ids into every response body,
|
||||||
|
#' which is the cost this issue exists to remove.
|
||||||
|
#' @noRd
|
||||||
|
.cohort_count <- function(con, cohort) {
|
||||||
|
sql <- sprintf(
|
||||||
|
"SELECT COUNT(*) AS n FROM canonical_fips_xwalk WHERE %s",
|
||||||
|
.cohort_sql(cohort)
|
||||||
|
)
|
||||||
|
as.integer(DBI::dbGetQuery(con, sql)$n[[1]])
|
||||||
|
}
|
||||||
|
|
||||||
|
#' The provenance `scope$cohort` block for a predicate cohort, or NULL when
|
||||||
|
#' the cohort was named by id alone (in which case `govids_found`/
|
||||||
|
#' `govids_missing` already describe it exactly).
|
||||||
|
#' @noRd
|
||||||
|
.cohort_provenance <- function(con, cohort) {
|
||||||
|
if (!.cohort_by_predicate(cohort)) return(NULL)
|
||||||
|
list(
|
||||||
|
state = if (is.null(cohort$state)) NA_character_ else as.character(cohort$state),
|
||||||
|
type = if (is.null(cohort$type)) NA_character_ else as.character(cohort$type),
|
||||||
|
n_governments = .cohort_count(con, cohort)
|
||||||
|
)
|
||||||
|
}
|
||||||
+149
@@ -0,0 +1,149 @@
|
|||||||
|
# R/complete.R
|
||||||
|
#
|
||||||
|
# `complete = TRUE` on the money verbs. Fills the requested grid so that a
|
||||||
|
# cell the corpus does not carry still appears, labelled with WHY it is
|
||||||
|
# missing.
|
||||||
|
#
|
||||||
|
# The corpus stopped storing the wide era's explicit zeros
|
||||||
|
# (cog_pipeline#64, series break SB194), which made absence ambiguous:
|
||||||
|
#
|
||||||
|
# <= FY2011 dense_source absent => Census published $0 (census_zero)
|
||||||
|
# >= FY2012 sparse_source absent => not reported, unknown (not_reported)
|
||||||
|
#
|
||||||
|
# Before sparsification a wide-era query whose cells were all $0 came back as
|
||||||
|
# explicit $0 rows; afterwards it came back empty, with nothing to say which
|
||||||
|
# of the two meanings applied. This restores that -- and improves on it,
|
||||||
|
# because the pre-sparsification corpus could not distinguish the two either.
|
||||||
|
#
|
||||||
|
# `census_zero` fills carry `amt_nominal = 0`; `not_reported` fills carry NA.
|
||||||
|
# That difference is the entire point: writing 0 into a modern absence would
|
||||||
|
# invent data, which is the error the representation contract exists to stop.
|
||||||
|
|
||||||
|
#' @noRd
|
||||||
|
.abort_complete_unsupported <- function(reason, alternative) {
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"{.code complete = TRUE} is not supported for this query.",
|
||||||
|
x = reason,
|
||||||
|
i = alternative
|
||||||
|
), class = "uscogdata_complete_unsupported")
|
||||||
|
}
|
||||||
|
|
||||||
|
#' @noRd
|
||||||
|
.require_representation <- function(con, manifest) {
|
||||||
|
needed <- c("representation.parquet", "code_set.parquet")
|
||||||
|
missing <- needed[!vapply(needed, function(f) .corpus_has_table(manifest, f),
|
||||||
|
logical(1))]
|
||||||
|
if (length(missing) == 0L) return(invisible(TRUE))
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"This corpus does not publish the representation contract.",
|
||||||
|
x = "Missing: {.file {missing}}.",
|
||||||
|
i = "{.code complete = TRUE} needs those tables to know whether an absent cell means Census published $0 or means the government did not report.",
|
||||||
|
i = "They ship with corpora published from 2026-07-29 onward; re-point {.envvar USCOGDATA_URL} at a current corpus, or omit {.code complete}."
|
||||||
|
), class = "uscogdata_representation_unavailable")
|
||||||
|
}
|
||||||
|
|
||||||
|
#' The cells a government-year COULD carry: every code in force for that
|
||||||
|
#' government's own type, mapped through `summary_categories`, restricted to
|
||||||
|
#' the calling verb's crosswalk subtype scope (the same subtype-membership
|
||||||
|
#' classification the verb SQL itself uses -- e.g. the `primary` concept's
|
||||||
|
#' operations/capital/assistance) and (when given) its category filter.
|
||||||
|
#'
|
||||||
|
#' Scoped by `govs_type` deliberately. Filling against the union of all types
|
||||||
|
#' would invent cells that the government can never report -- a county row for
|
||||||
|
#' "state IG transfer to school districts" -- and those inventions would then
|
||||||
|
#' be indistinguishable from real census zeros.
|
||||||
|
#'
|
||||||
|
#' `NOT cs.is_aggregate` mirrors `spending_long` / `revenue_long`, which drop
|
||||||
|
#' aggregate rows. Without it the grid would offer cells the verb structurally
|
||||||
|
#' never returns, so every one of them would fill as a phantom $0.
|
||||||
|
#' @noRd
|
||||||
|
.completion_grid_sql <- function(subtype_col, cohort, years, category,
|
||||||
|
subtype_scope) {
|
||||||
|
category_pred <- if (is.null(category)) {
|
||||||
|
""
|
||||||
|
} else {
|
||||||
|
sprintf("AND c.category IN (%s)", .sql_lit_chr(category))
|
||||||
|
}
|
||||||
|
sprintf(
|
||||||
|
"SELECT DISTINCT
|
||||||
|
cs.year,
|
||||||
|
x.canonical_govid,
|
||||||
|
x.gov_name,
|
||||||
|
c.%1$s AS subtype_value,
|
||||||
|
c.category,
|
||||||
|
r.absence_means
|
||||||
|
FROM code_set cs
|
||||||
|
JOIN canonical_fips_xwalk x ON x.govs_type = cs.type
|
||||||
|
JOIN summary_categories c ON c.item_code = cs.item_code
|
||||||
|
JOIN representation r ON r.year = cs.year
|
||||||
|
WHERE %2$s
|
||||||
|
AND cs.year IN (%3$s)
|
||||||
|
AND NOT cs.is_aggregate
|
||||||
|
AND c.category IS NOT NULL
|
||||||
|
AND c.%1$s IN (%4$s)
|
||||||
|
%5$s",
|
||||||
|
subtype_col, .cohort_sql(cohort, "x.canonical_govid"),
|
||||||
|
paste(as.integer(years), collapse = ","),
|
||||||
|
.sql_lit_chr(subtype_scope), category_pred
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Fill `result` out to the full grid, stamping `value_source` on every row.
|
||||||
|
#'
|
||||||
|
#' Returns the completed tibble with a `.completion` attribute carrying the
|
||||||
|
#' provenance block. Reported rows are passed through untouched -- filling
|
||||||
|
#' must never alter or drop what the corpus actually published.
|
||||||
|
#' @noRd
|
||||||
|
.complete_result <- function(result, con, subtype_col, cohort, years, category,
|
||||||
|
subtype_scope) {
|
||||||
|
grid <- tibble::as_tibble(DBI::dbGetQuery(
|
||||||
|
con, .completion_grid_sql(subtype_col, cohort, years, category, subtype_scope)
|
||||||
|
))
|
||||||
|
|
||||||
|
result$value_source <- rep("reported", nrow(result))
|
||||||
|
if (nrow(grid) == 0L) {
|
||||||
|
attr(result, ".completion") <- list(
|
||||||
|
applied = TRUE, rows_filled = 0L, absence_means = list()
|
||||||
|
)
|
||||||
|
return(result)
|
||||||
|
}
|
||||||
|
|
||||||
|
names(grid)[names(grid) == "subtype_value"] <- subtype_col
|
||||||
|
key <- function(d) {
|
||||||
|
paste(d$year, d$canonical_govid, d[[subtype_col]], d$category, sep = "\r")
|
||||||
|
}
|
||||||
|
missing <- grid[!key(grid) %in% key(result), , drop = FALSE]
|
||||||
|
|
||||||
|
if (nrow(missing) > 0L) {
|
||||||
|
filled <- tibble::tibble(
|
||||||
|
year = as.integer(missing$year),
|
||||||
|
canonical_govid = as.character(missing$canonical_govid),
|
||||||
|
gov_name = as.character(missing$gov_name),
|
||||||
|
category = as.character(missing$category),
|
||||||
|
# census_zero is a value Census published; not_reported is unknown and
|
||||||
|
# must stay NA. Collapsing the two to 0 is the defect, not the fill.
|
||||||
|
amt_nominal = ifelse(missing$absence_means == "census_zero",
|
||||||
|
0, NA_real_),
|
||||||
|
codes_included = NA_character_,
|
||||||
|
aggregate_fallback = NA,
|
||||||
|
value_source = as.character(missing$absence_means)
|
||||||
|
)
|
||||||
|
filled[[subtype_col]] <- as.character(missing[[subtype_col]])
|
||||||
|
if ("notes" %in% names(result)) filled$notes <- NA_character_
|
||||||
|
|
||||||
|
result <- dplyr::bind_rows(result, filled)
|
||||||
|
result <- result[order(result$year, result$canonical_govid,
|
||||||
|
result[[subtype_col]], result$category), ,
|
||||||
|
drop = FALSE]
|
||||||
|
}
|
||||||
|
|
||||||
|
rules <- unique(grid[, c("year", "absence_means")])
|
||||||
|
attr(result, ".completion") <- list(
|
||||||
|
applied = TRUE,
|
||||||
|
rows_filled = nrow(missing),
|
||||||
|
absence_means = stats::setNames(
|
||||||
|
as.list(as.character(rules$absence_means)), as.character(rules$year)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
result
|
||||||
|
}
|
||||||
+96
-3
@@ -5,9 +5,28 @@
|
|||||||
.uscogdata_env <- new.env(parent = emptyenv())
|
.uscogdata_env <- new.env(parent = emptyenv())
|
||||||
|
|
||||||
.uscogdata_defaults <- list(
|
.uscogdata_defaults <- list(
|
||||||
url = "https://cloud.civilytics.org/s/REPLACE_WITH_SHARE_TOKEN/download/",
|
# Public HuggingFace mirror of the published corpus: CC-BY-4.0, no
|
||||||
|
# credential, CDN-backed. This is the default so `library(uscogdata)`
|
||||||
|
# followed by a verb works with zero configuration -- previously the
|
||||||
|
# default was a REPLACE_WITH_SHARE_TOKEN sentinel and no document in the
|
||||||
|
# package supplied a working URL, so a new user had no path to a session.
|
||||||
|
#
|
||||||
|
# The trailing slash is required: every consumer concatenates onto this
|
||||||
|
# (see .resolve_url(), which enforces it anyway).
|
||||||
|
#
|
||||||
|
# Override with USCOGDATA_URL or options(uscogdata.url=) to read a
|
||||||
|
# Nextcloud share or a local copy made by cog_mirror().
|
||||||
|
url = "https://huggingface.co/datasets/civilytics/us-cog-finance/resolve/main/",
|
||||||
cache_dir = NULL,
|
cache_dir = NULL,
|
||||||
manifest_ttl_secs = 3600L
|
manifest_ttl_secs = 3600L,
|
||||||
|
# NULL means "emit no pragma", which leaves DuckDB's own defaults intact:
|
||||||
|
# every visible core, and 80% of RAM. That is right for one interactive
|
||||||
|
# session on a dedicated machine and wrong for a server, where several
|
||||||
|
# readers share a box and each would otherwise claim all of it. See
|
||||||
|
# .resolve_duckdb_threads() for why this is a supported option rather than
|
||||||
|
# something a consumer reaches into the namespace to set.
|
||||||
|
duckdb_threads = NULL,
|
||||||
|
duckdb_memory_limit = NULL
|
||||||
)
|
)
|
||||||
|
|
||||||
#' Resolve a config value: env var > option > default
|
#' Resolve a config value: env var > option > default
|
||||||
@@ -21,9 +40,83 @@
|
|||||||
.uscogdata_defaults[[key]]
|
.uscogdata_defaults[[key]]
|
||||||
}
|
}
|
||||||
|
|
||||||
.resolve_url <- function() .cfg("url")
|
#' Resolve the corpus URL, guaranteeing the trailing slash the package assumes.
|
||||||
|
#'
|
||||||
|
#' Every consumer builds locations by CONCATENATION -- `paste0(url,
|
||||||
|
#' "manifest.json")` in manifest.R, `paste0(url, e$path)` in mirror.R, and the
|
||||||
|
#' parquet glob in views.R -- and mirror.R:104 documents the invariant outright
|
||||||
|
#' ('url ends in "/"'). Nothing enforced it, so a URL entered without the slash
|
||||||
|
#' failed silently and misleadingly:
|
||||||
|
#'
|
||||||
|
#' HTTPS -> ".../downloadmanifest.json"; the host answers with an HTML 404
|
||||||
|
#' page, which lands in the JSON parser as the lexical error
|
||||||
|
#' reported in issue #3 -- pointing the user at "login page / wrong
|
||||||
|
#' share" when the real cause was one missing character.
|
||||||
|
#' local -> ".../corpusdata/long/**/*.parquet" and a DuckDB "No files found".
|
||||||
|
#'
|
||||||
|
#' Normalizing here fixes every consumer at once, rather than each call site
|
||||||
|
#' re-deriving the same invariant. An empty setting is passed through
|
||||||
|
#' untouched so manifest.R's "not configured" guard still fires instead of the
|
||||||
|
#' value degrading into a bare "/" filesystem root.
|
||||||
|
#' @noRd
|
||||||
|
.resolve_url <- function() {
|
||||||
|
url <- .cfg("url")
|
||||||
|
if (is.null(url) || !nzchar(url) || grepl("/$", url)) return(url)
|
||||||
|
paste0(url, "/")
|
||||||
|
}
|
||||||
|
|
||||||
.resolve_cache_dir <- function() {
|
.resolve_cache_dir <- function() {
|
||||||
v <- .cfg("cache_dir")
|
v <- .cfg("cache_dir")
|
||||||
if (is.null(v)) tools::R_user_dir("uscogdata", "cache") else v
|
if (is.null(v)) tools::R_user_dir("uscogdata", "cache") else v
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#' Resolve the DuckDB thread cap, or NULL to leave DuckDB's default alone.
|
||||||
|
#'
|
||||||
|
#' `cog_open()` used to connect with a bare `dbConnect()` and set no `threads`
|
||||||
|
#' pragma, so DuckDB claimed every core it could see. cog-api works around that
|
||||||
|
#' by reaching into this namespace at boot --
|
||||||
|
#' `getFromNamespace(".ensure_session", "uscogdata")()` followed by a manual
|
||||||
|
#' `SET threads` -- which depends on a private name and on the session already
|
||||||
|
#' being open. Making it a resolved option removes the reason to do that.
|
||||||
|
#'
|
||||||
|
#' `.cfg()` returns an environment variable as CHARACTER, so this coerces
|
||||||
|
#' rather than trusting the type: `USCOGDATA_DUCKDB_THREADS=4` arrives as "4",
|
||||||
|
#' and `sprintf("SET threads TO %d", "4")` would abort inside the connection
|
||||||
|
#' path with an error about the pragma rather than about the setting.
|
||||||
|
#' @noRd
|
||||||
|
.resolve_duckdb_threads <- function() {
|
||||||
|
v <- .cfg("duckdb_threads")
|
||||||
|
if (is.null(v) || (is.character(v) && !nzchar(v))) return(NULL)
|
||||||
|
n <- suppressWarnings(as.integer(v))
|
||||||
|
if (length(n) != 1L || is.na(n) || n < 1L) {
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"{.envvar USCOGDATA_DUCKDB_THREADS} must be a single positive integer.",
|
||||||
|
x = "Got {.val {v}}.",
|
||||||
|
i = "Unset it (or {.code options(uscogdata.duckdb_threads = NULL)}) to use DuckDB's default of every visible core."
|
||||||
|
), class = "uscogdata_invalid_duckdb_threads")
|
||||||
|
}
|
||||||
|
n
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Resolve the DuckDB memory limit, or NULL to leave DuckDB's default alone.
|
||||||
|
#'
|
||||||
|
#' The value is a DuckDB size string (`"4GB"`, `"512MB"`). Only its SHAPE is
|
||||||
|
#' checked here -- DuckDB owns the unit vocabulary, and re-implementing that
|
||||||
|
#' parse would be a second definition free to drift from the engine's. An
|
||||||
|
#' unrecognised unit therefore surfaces as DuckDB's own error at `SET` time,
|
||||||
|
#' which names the setting correctly; the check here exists to reject the
|
||||||
|
#' inputs that would otherwise reach the connection as a SQL fragment.
|
||||||
|
#' @noRd
|
||||||
|
.resolve_duckdb_memory_limit <- function() {
|
||||||
|
v <- .cfg("duckdb_memory_limit")
|
||||||
|
if (is.null(v) || (is.character(v) && !nzchar(v))) return(NULL)
|
||||||
|
if (length(v) != 1L || !is.character(v) ||
|
||||||
|
!grepl("^[0-9]+(\\.[0-9]+)?\\s*[A-Za-z]{0,3}$", v)) {
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"{.envvar USCOGDATA_DUCKDB_MEMORY_LIMIT} must be a single DuckDB size string.",
|
||||||
|
x = "Got {.val {v}}.",
|
||||||
|
i = "Examples: {.val 4GB}, {.val 512MB}, {.val 1.5GB}."
|
||||||
|
), class = "uscogdata_invalid_duckdb_memory_limit")
|
||||||
|
}
|
||||||
|
trimws(v)
|
||||||
|
}
|
||||||
|
|||||||
+107
@@ -0,0 +1,107 @@
|
|||||||
|
# R/coverage.R
|
||||||
|
#
|
||||||
|
# Reporting-coverage disclosure for the multi-government verbs (uscogdata#13,
|
||||||
|
# findings F-020 and F-023).
|
||||||
|
#
|
||||||
|
# The Census of Governments is a COMPLETE CENSUS only in years ending in 2 and
|
||||||
|
# 7. Every other year is a sample, and the sample varies enormously: on the
|
||||||
|
# bundled fixture, Wisconsin's 608-city universe reports 597 governments in
|
||||||
|
# FY2012 and 112 in FY2019. Summing "whatever reported" across those years is
|
||||||
|
# what the verbs have always done -- correctly -- but the return value said
|
||||||
|
# nothing about it, so a statewide total resting on 18% of the universe looked
|
||||||
|
# exactly like one resting on 98%.
|
||||||
|
#
|
||||||
|
# Owner's settled design: a `coverage` argument selecting WHICH units to
|
||||||
|
# include, plus always-on metadata saying how many there were either way. The
|
||||||
|
# principle behind it: using these verbs correctly must not require the caller
|
||||||
|
# to know the survey calendar.
|
||||||
|
|
||||||
|
# Years ending in 2 or 7 are full censuses of every government; all others are
|
||||||
|
# samples.
|
||||||
|
.CENSUS_YEAR_ENDINGS <- c(2L, 7L)
|
||||||
|
|
||||||
|
#' @noRd
|
||||||
|
.is_census_year <- function(years) {
|
||||||
|
as.integer(years) %% 10L %in% .CENSUS_YEAR_ENDINGS
|
||||||
|
}
|
||||||
|
|
||||||
|
#' @noRd
|
||||||
|
.validate_coverage <- function(coverage) {
|
||||||
|
tryCatch(
|
||||||
|
match.arg(coverage, c("all", "census", "consistent")),
|
||||||
|
error = function(e) {
|
||||||
|
cli::cli_abort(
|
||||||
|
"`coverage` must be one of {.val all}, {.val census} or {.val consistent}.",
|
||||||
|
class = "uscogdata_invalid_coverage", parent = e
|
||||||
|
)
|
||||||
|
}
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Restrict `years` to census years for `coverage = "census"`.
|
||||||
|
#'
|
||||||
|
#' Aborts rather than returning an empty result when the requested range holds
|
||||||
|
#' no census year: silently handing back zero rows for a query the caller
|
||||||
|
#' believes they made is the failure mode this whole issue is about.
|
||||||
|
#' @noRd
|
||||||
|
.apply_census_years <- function(years, coverage, verb) {
|
||||||
|
if (!identical(coverage, "census")) return(as.integer(years))
|
||||||
|
keep <- as.integer(years)[.is_census_year(years)]
|
||||||
|
if (length(keep) == 0L) {
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"{.code coverage = \"census\"} leaves no years to query.",
|
||||||
|
x = "None of the requested years end in 2 or 7: {.val {sort(unique(as.integer(years)))}}.",
|
||||||
|
i = "Census of Governments years ending in 2 or 7 are complete censuses; all others are samples.",
|
||||||
|
i = "Use {.code coverage = \"all\"} (the default) to keep every requested year, or request a census year."
|
||||||
|
), class = "uscogdata_no_census_years")
|
||||||
|
}
|
||||||
|
sort(keep)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Keep only units that report in EVERY requested year (a balanced panel).
|
||||||
|
#'
|
||||||
|
#' `id_col` is the government identifier; `keep_ids` are rows exempt from the
|
||||||
|
#' filter (the peer-comparison target, which is the subject of the comparison
|
||||||
|
#' rather than a member of the cohort being balanced).
|
||||||
|
#' @noRd
|
||||||
|
.filter_consistent <- function(result, years, id_col = "canonical_govid",
|
||||||
|
keep_ids = character(0)) {
|
||||||
|
years <- unique(as.integer(years))
|
||||||
|
if (nrow(result) == 0L || length(years) <= 1L) return(result)
|
||||||
|
ids <- setdiff(unique(result[[id_col]]), c(NA, keep_ids))
|
||||||
|
present <- vapply(ids, function(g) {
|
||||||
|
all(years %in% unique(as.integer(result$year[result[[id_col]] == g])))
|
||||||
|
}, logical(1))
|
||||||
|
consistent <- c(ids[present], keep_ids)
|
||||||
|
result[result[[id_col]] %in% consistent | is.na(result[[id_col]]), ,
|
||||||
|
drop = FALSE]
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Per-year coverage metadata, always attached regardless of mode.
|
||||||
|
#'
|
||||||
|
#' Built from the REQUESTED years rather than the years present in the result,
|
||||||
|
#' so a year in which nothing reported still appears -- with
|
||||||
|
#' `n_units_reporting = 0`, which is precisely the disclosure a silently
|
||||||
|
#' missing year fails to make.
|
||||||
|
#'
|
||||||
|
#' `n_units_reporting` describes the result the caller actually received, so
|
||||||
|
#' under `coverage = "consistent"` it reports the balanced count. `is_census_year`
|
||||||
|
#' is a statement about the SURVEY CALENDAR, never a claim of completeness:
|
||||||
|
#' FY1967 is a census year in which only 97 of Wisconsin's 608 cities report.
|
||||||
|
#' `n_units_reporting` is the number that tells the truth.
|
||||||
|
#' @noRd
|
||||||
|
.coverage_table <- function(result, years, n_expected,
|
||||||
|
id_col = "canonical_govid", rows = NULL) {
|
||||||
|
years <- sort(unique(as.integer(years)))
|
||||||
|
src <- if (is.null(rows)) result else rows
|
||||||
|
reporting <- vapply(years, function(y) {
|
||||||
|
ids <- src[[id_col]][as.integer(src$year) == y]
|
||||||
|
length(unique(ids[!is.na(ids)]))
|
||||||
|
}, integer(1))
|
||||||
|
tibble::tibble(
|
||||||
|
year = years,
|
||||||
|
n_units_reporting = as.integer(reporting),
|
||||||
|
n_units_expected = rep(as.integer(n_expected), length(years)),
|
||||||
|
is_census_year = .is_census_year(years)
|
||||||
|
)
|
||||||
|
}
|
||||||
+213
@@ -11,6 +11,30 @@
|
|||||||
#' returns `result` invisibly for chaining. `"list"` returns the raw
|
#' returns `result` invisibly for chaining. `"list"` returns the raw
|
||||||
#' provenance list (identical to `attr(result, "provenance")`).
|
#' provenance list (identical to `attr(result, "provenance")`).
|
||||||
#' @return Either `result` (invisibly) or the provenance list.
|
#' @return Either `result` (invisibly) or the provenance list.
|
||||||
|
#' @section Two kinds of series break:
|
||||||
|
#' Catalogued breaks reach you without being asked for, in two disjoint
|
||||||
|
#' fields, because a caveat about one series and a caveat about the whole
|
||||||
|
#' corpus are different claims:
|
||||||
|
#'
|
||||||
|
#' * **`series_break_refs`** — breaks matched against the item codes actually
|
||||||
|
#' present in this result. A break in one code you queried.
|
||||||
|
#' * **`corpus_break_refs`** — breaks catalogued with `fin_code = "ALL"`,
|
||||||
|
#' which are statements about the corpus rather than about any one code:
|
||||||
|
#' dollar precision across the 1976/1977 boundary (`SB085`), imputation
|
||||||
|
#' exclusion from FY2002 (`SB087`), the FY2012 dense-to-sparse
|
||||||
|
#' representation change (`SB194`), and the FY2017 government-identifier
|
||||||
|
#' change (`SB086`). These are selected on the break-year window alone.
|
||||||
|
#'
|
||||||
|
#' `SB194` is the one most likely to matter: a query spanning FY2011 to FY2012
|
||||||
|
#' crosses the boundary where an absent cell stops meaning "Census published
|
||||||
|
#' $0" and starts meaning "not reported".
|
||||||
|
#' @section Other provenance blocks:
|
||||||
|
#' `transformations$units_conversion` records the `$1,000s`-to-dollars
|
||||||
|
#' multiply that every amount column has already had applied.
|
||||||
|
#' `transformations$per_capita` records the population denominator and its
|
||||||
|
#' year range. `coverage` and `coverage_mode` appear on multi-government
|
||||||
|
#' results (see [cog_geographic_rollup()]). `completion` appears when
|
||||||
|
#' `complete = TRUE`. `balance_caveats` appears on [cog_balances()] results.
|
||||||
#' @export
|
#' @export
|
||||||
cog_explain <- function(result, format = c("print", "list")) {
|
cog_explain <- function(result, format = c("print", "list")) {
|
||||||
format <- match.arg(format)
|
format <- match.arg(format)
|
||||||
@@ -51,6 +75,43 @@ cog_explain <- function(result, format = c("print", "list")) {
|
|||||||
cli::cli_text("Category: (all)")
|
cli::cli_text("Category: (all)")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (!is.null(prov$basis)) {
|
||||||
|
note <- if (!is.null(prov$basis_note) && !is.na(prov$basis_note)) {
|
||||||
|
sprintf(" (%s)", prov$basis_note)
|
||||||
|
} else {
|
||||||
|
""
|
||||||
|
}
|
||||||
|
cli::cli_text("Basis: {prov$basis}{note}")
|
||||||
|
}
|
||||||
|
|
||||||
|
# Each verb reports its OWN concept. Both fields are always present (each
|
||||||
|
# defaults to its concept's default), so printing `expenditure_concept`
|
||||||
|
# unconditionally would tell a cog_revenue() caller "Concept: primary",
|
||||||
|
# which names a spending concept their result has nothing to do with.
|
||||||
|
if (identical(prov$verb, "cog_revenue")) {
|
||||||
|
if (!is.null(prov$revenue_concept)) {
|
||||||
|
cli::cli_text("Concept: {prov$revenue_concept} revenue")
|
||||||
|
}
|
||||||
|
} else if (identical(prov$verb, "cog_balances")) {
|
||||||
|
# Both concept fields are deliberately NA here (a stock has no flow
|
||||||
|
# concept). Printing the raw NA reads as a missing value rather than an
|
||||||
|
# intentional one, so say what it means instead.
|
||||||
|
cli::cli_text("Concept: not applicable (holdings are a stock, not a flow)")
|
||||||
|
} else if (!is.null(prov$expenditure_concept)) {
|
||||||
|
concept_note <- if (!is.null(prov$expenditure_concept_note) &&
|
||||||
|
!is.na(prov$expenditure_concept_note)) {
|
||||||
|
sprintf(" (%s)", prov$expenditure_concept_note)
|
||||||
|
} else {
|
||||||
|
""
|
||||||
|
}
|
||||||
|
cli::cli_text("Concept: {prov$expenditure_concept}{concept_note}")
|
||||||
|
if (isTRUE(prov$expenditure_concept_direct_suppressed)) {
|
||||||
|
cli::cli_alert_warning(
|
||||||
|
"Direct leg unavailable for at least one requested (year, category) -- affected rows report intergovernmental dollars alone, not Direct + IG. See each row's notes."
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
cli::cli_h2("Codes observed")
|
cli::cli_h2("Codes observed")
|
||||||
codes <- prov$codes_summed$observed
|
codes <- prov$codes_summed$observed
|
||||||
if (length(codes) == 0L) {
|
if (length(codes) == 0L) {
|
||||||
@@ -66,6 +127,114 @@ cog_explain <- function(result, format = c("print", "list")) {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
h <- prov$harmonization
|
||||||
|
if (!is.null(h) && isTRUE(h$applied)) {
|
||||||
|
cli::cli_h2("Harmonization")
|
||||||
|
cli::cli_text(
|
||||||
|
"Excluded {h$na_rows_excluded} row(s) with no harmonized_code (${format(h$na_amount_excluded, big.mark = ',')})"
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
rc <- prov$recipe
|
||||||
|
if (!is.null(rc)) {
|
||||||
|
cli::cli_h2("Recipe")
|
||||||
|
cli::cli_text("{rc$recipe_id}: {rc$label}")
|
||||||
|
comp_lines <- vapply(rc$components, function(x) {
|
||||||
|
sprintf("%s (%s, %s-%s, weight=%s)", x$component_code, x$gov_type_scope,
|
||||||
|
x$year_min, x$year_max, x$weight)
|
||||||
|
}, character(1))
|
||||||
|
cli::cli_ul(comp_lines)
|
||||||
|
}
|
||||||
|
|
||||||
|
if (length(prov$suggestions) > 0L) {
|
||||||
|
cli::cli_h2("Suggestions")
|
||||||
|
sugg_lines <- vapply(prov$suggestions, function(s) {
|
||||||
|
line <- sprintf("%s -- %s (years %s-%s): %s", s$recipe_id, s$label,
|
||||||
|
s$available_years[1], s$available_years[2], s$hint)
|
||||||
|
if (isTRUE(s$suppressed_amount > 0)) {
|
||||||
|
line <- paste0(line, sprintf(" [$%s excluded from %s: %s]",
|
||||||
|
formatC(s$suppressed_amount, format = "f", digits = 0, big.mark = ","),
|
||||||
|
paste0("FY", s$suppressed_years, collapse = ", "),
|
||||||
|
paste(s$suppressed_codes, collapse = ", ")))
|
||||||
|
}
|
||||||
|
line
|
||||||
|
}, character(1))
|
||||||
|
cli::cli_ul(sugg_lines)
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!is.null(prov$coverage) && nrow(prov$coverage) > 0L) {
|
||||||
|
cli::cli_h2("Reporting coverage")
|
||||||
|
cli::cli_text("Mode: {prov$coverage_mode %||% 'all'}")
|
||||||
|
cov <- prov$coverage
|
||||||
|
cli::cli_ul(sprintf(
|
||||||
|
"%d: %d of %d units reporting (%.0f%%) -- %s year",
|
||||||
|
cov$year, cov$n_units_reporting, cov$n_units_expected,
|
||||||
|
100 * cov$n_units_reporting / pmax(cov$n_units_expected, 1L),
|
||||||
|
ifelse(cov$is_census_year, "census", "sample")
|
||||||
|
))
|
||||||
|
if (any(!cov$is_census_year)) {
|
||||||
|
cli::cli_text(
|
||||||
|
"Note: the Census of Governments is a complete census only in years ending in 2 or 7; every other year is a sample."
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (isTRUE(prov$completion$applied)) {
|
||||||
|
cli::cli_h2("Completion")
|
||||||
|
cli::cli_text(
|
||||||
|
"Filled {prov$completion$rows_filled} absent cell(s) from the corpus code set."
|
||||||
|
)
|
||||||
|
rules <- prov$completion$absence_means
|
||||||
|
if (length(rules) > 0L) {
|
||||||
|
cli::cli_ul(vapply(names(rules), function(y) {
|
||||||
|
sprintf("%s: an absent cell means %s", y,
|
||||||
|
if (identical(rules[[y]], "census_zero")) {
|
||||||
|
"Census published $0 (filled as 0)"
|
||||||
|
} else {
|
||||||
|
"the government did not report (filled as NA, not 0)"
|
||||||
|
})
|
||||||
|
}, character(1)))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (length(prov$series_break_refs) > 0L) {
|
||||||
|
cli::cli_h2("Series breaks")
|
||||||
|
cli::cli_ul(.series_break_story_lines(prov$series_break_refs))
|
||||||
|
}
|
||||||
|
|
||||||
|
# Kept in a section of its own: these qualify the whole result, so folding
|
||||||
|
# them in with the per-code breaks above would invite reading them as a
|
||||||
|
# caveat about one series.
|
||||||
|
if (length(prov$corpus_break_refs) > 0L) {
|
||||||
|
cli::cli_h2("Corpus-wide caveats")
|
||||||
|
cli::cli_ul(.series_break_story_lines(prov$corpus_break_refs))
|
||||||
|
}
|
||||||
|
|
||||||
|
# Balance results only (NULL on money-verb provenance, so they are
|
||||||
|
# unaffected). This is the ONLY on-demand surface for the GAAP disclosure:
|
||||||
|
# .emit_balance_caveats() fires at most once per session, and is routinely
|
||||||
|
# consumed by a suppressMessages() call or by a knitted chunk nobody reads,
|
||||||
|
# so a caller who deliberately audits a result with cog_explain() must still
|
||||||
|
# be told.
|
||||||
|
bc <- prov$balance_caveats
|
||||||
|
if (!is.null(bc)) {
|
||||||
|
cli::cli_h2("Holdings caveats")
|
||||||
|
if (!is.null(bc$not_gaap_note)) cli::cli_alert_warning(bc$not_gaap_note)
|
||||||
|
if (length(bc$truncated) > 0L) {
|
||||||
|
cli::cli_text(
|
||||||
|
"Requested years extend beyond what these families actually cover:"
|
||||||
|
)
|
||||||
|
cli::cli_ul(vapply(bc$truncated, function(s) {
|
||||||
|
w <- bc$coverage_window[[s]]
|
||||||
|
if (length(w) == 2L) {
|
||||||
|
sprintf("%s: covered %s-%s in this corpus", s, w[1], w[2])
|
||||||
|
} else {
|
||||||
|
s
|
||||||
|
}
|
||||||
|
}, character(1)))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
cli::cli_h2("Transformations")
|
cli::cli_h2("Transformations")
|
||||||
uc <- prov$transformations$units_conversion
|
uc <- prov$transformations$units_conversion
|
||||||
if (isTRUE(uc$applied)) {
|
if (isTRUE(uc$applied)) {
|
||||||
@@ -74,6 +243,16 @@ cog_explain <- function(result, format = c("print", "list")) {
|
|||||||
pc <- prov$transformations$per_capita
|
pc <- prov$transformations$per_capita
|
||||||
if (isTRUE(pc$applied)) {
|
if (isTRUE(pc$applied)) {
|
||||||
cli::cli_text("Per-capita denominator: {pc$denominator_source}")
|
cli::cli_text("Per-capita denominator: {pc$denominator_source}")
|
||||||
|
if (length(pc$popyear_range) == 2L) {
|
||||||
|
lo <- .expand_popyear(pc$popyear_range[1])
|
||||||
|
hi <- .expand_popyear(pc$popyear_range[2])
|
||||||
|
cli::cli_text(" popyear range: {lo}-{hi}")
|
||||||
|
}
|
||||||
|
if (!is.null(pc$pop_source_counts)) {
|
||||||
|
cli::cli_text(
|
||||||
|
" pop_source counts: census_f33={pc$pop_source_counts$census_f33}, unavailable={pc$pop_source_counts$unavailable}"
|
||||||
|
)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
infl <- prov$transformations$inflation
|
infl <- prov$transformations$inflation
|
||||||
if (isTRUE(infl$applied)) {
|
if (isTRUE(infl$applied)) {
|
||||||
@@ -98,3 +277,37 @@ cog_explain <- function(result, format = c("print", "list")) {
|
|||||||
|
|
||||||
invisible(NULL)
|
invisible(NULL)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# One "break-story" line per referenced break_id: "SB109 (2005): <join_advice>".
|
||||||
|
# Re-queries series_breaks_pq for the detail (break_year, join_advice) that
|
||||||
|
# provenance$series_break_refs deliberately doesn't carry (the schema keeps
|
||||||
|
# that field to a plain id array). Falls back to bare ids if no session is
|
||||||
|
# available (e.g. explaining a result after cog_close()) rather than
|
||||||
|
# erroring cog_explain() over a cosmetic detail.
|
||||||
|
#' @noRd
|
||||||
|
.series_break_story_lines <- function(break_ids) {
|
||||||
|
con <- tryCatch(.ensure_session(), error = function(e) NULL)
|
||||||
|
if (is.null(con) || !DBI::dbIsValid(con)) return(break_ids)
|
||||||
|
detail <- tryCatch(
|
||||||
|
DBI::dbGetQuery(con, sprintf(
|
||||||
|
"SELECT break_id, break_year, join_advice FROM series_breaks_pq
|
||||||
|
WHERE break_id IN (%s) ORDER BY break_id",
|
||||||
|
.sql_lit_chr(break_ids)
|
||||||
|
)),
|
||||||
|
error = function(e) NULL
|
||||||
|
)
|
||||||
|
if (is.null(detail) || nrow(detail) == 0L) return(break_ids)
|
||||||
|
sprintf("%s (%s): %s", detail$break_id, detail$break_year, detail$join_advice)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Expand a 2-digit Census popyear (e.g. 19) to a 4-digit calendar year (2019).
|
||||||
|
# F-33 metadata stores popyear as 2 digits; pivot at 70 to handle a future
|
||||||
|
# corpus that ever spans pre-1970 vintages, though current scope is 2000+.
|
||||||
|
#' @noRd
|
||||||
|
.expand_popyear <- function(yy) {
|
||||||
|
yy <- as.integer(yy)
|
||||||
|
if (length(yy) == 0L || is.na(yy)) return(NA_integer_)
|
||||||
|
if (yy >= 100L) return(yy) # already 4-digit
|
||||||
|
if (yy < 70L) return(2000L + yy)
|
||||||
|
1900L + yy
|
||||||
|
}
|
||||||
|
|||||||
+127
-10
@@ -1,5 +1,66 @@
|
|||||||
# R/manifest.R
|
# R/manifest.R
|
||||||
|
|
||||||
|
# Sentinel substring baked into the placeholder default URL. If we see this
|
||||||
|
# in the resolved URL, the user hasn't configured USCOGDATA_URL yet.
|
||||||
|
.PLACEHOLDER_TOKEN <- "REPLACE_WITH_SHARE_TOKEN"
|
||||||
|
|
||||||
|
#' Abort with actionable guidance when the resolved corpus URL is still the
|
||||||
|
#' placeholder shipped with the package (or any URL containing the sentinel).
|
||||||
|
#' Called from `cog_open()` before any I/O so users see a clear message
|
||||||
|
#' instead of a downstream JSON parse error.
|
||||||
|
#' @noRd
|
||||||
|
.check_url_configured <- function(url) {
|
||||||
|
if (!is.character(url) || length(url) != 1L || !nzchar(url)) {
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"USCOGDATA_URL is not configured.",
|
||||||
|
i = "Set the corpus location via one of:",
|
||||||
|
"*" = "{.code Sys.setenv(USCOGDATA_URL = \"<url-or-local-path>/\")}",
|
||||||
|
"*" = "{.code options(uscogdata.url = \"<url-or-local-path>/\")}",
|
||||||
|
i = "For an offline smoke test, use the bundled fixture: {.code system.file(\"extdata/fixture_corpus\", package = \"uscogdata\")}."
|
||||||
|
), class = "uscogdata_url_not_configured")
|
||||||
|
}
|
||||||
|
if (grepl(.PLACEHOLDER_TOKEN, url, fixed = TRUE)) {
|
||||||
|
sentinel <- .PLACEHOLDER_TOKEN
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"USCOGDATA_URL is not configured (placeholder URL detected).",
|
||||||
|
x = "Current value contains the sentinel {.val {sentinel}}: {.url {url}}",
|
||||||
|
i = "Set the corpus location via one of:",
|
||||||
|
"*" = "{.code Sys.setenv(USCOGDATA_URL = \"<url-or-local-path>/\")}",
|
||||||
|
"*" = "{.code options(uscogdata.url = \"<url-or-local-path>/\")}",
|
||||||
|
i = "For an offline smoke test, use the bundled fixture: {.code system.file(\"extdata/fixture_corpus\", package = \"uscogdata\")}.",
|
||||||
|
i = "The public corpus is the default: unset USCOGDATA_URL to use it, or point it at a local copy made by {.code cog_mirror()}."
|
||||||
|
), class = "uscogdata_url_not_configured")
|
||||||
|
}
|
||||||
|
invisible(url)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Try to parse a JSON file. Returns parsed object on success, NULL on
|
||||||
|
#' any parse failure (so callers can decide whether to refetch).
|
||||||
|
#' @noRd
|
||||||
|
.try_parse_manifest_file <- function(path) {
|
||||||
|
tryCatch(
|
||||||
|
jsonlite::fromJSON(path, simplifyVector = FALSE),
|
||||||
|
error = function(e) NULL
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Abort with a clear, classified error when a manifest payload (string or
|
||||||
|
#' file) cannot be parsed as JSON. Surfaces the URL, content-type if known,
|
||||||
|
#' and the underlying parse error.
|
||||||
|
#' @noRd
|
||||||
|
.abort_invalid_manifest <- function(source, content_type = NA_character_, parse_error = NULL) {
|
||||||
|
ct <- if (is.na(content_type) || !nzchar(content_type)) "<unknown>" else content_type
|
||||||
|
pmsg <- if (is.null(parse_error)) "" else conditionMessage(parse_error)
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"Corpus manifest is not valid JSON.",
|
||||||
|
x = "Source: {source}",
|
||||||
|
i = "Content-Type: {ct}",
|
||||||
|
i = "Likely causes: USCOGDATA_URL points at a login page, a 404 HTML page, or the wrong share; or the corpus has not been published yet.",
|
||||||
|
i = "Set USCOGDATA_URL to a directory (local path or HTTPS) that serves manifest.json directly.",
|
||||||
|
if (nzchar(pmsg)) c(">" = "Parse error: {pmsg}") else NULL
|
||||||
|
), class = "uscogdata_invalid_manifest")
|
||||||
|
}
|
||||||
|
|
||||||
#' Fetch manifest.json from URL (or read from a local fixture path),
|
#' Fetch manifest.json from URL (or read from a local fixture path),
|
||||||
#' cache locally, validate TTL.
|
#' cache locally, validate TTL.
|
||||||
#' @noRd
|
#' @noRd
|
||||||
@@ -11,23 +72,54 @@
|
|||||||
if (!file.exists(local_manifest)) {
|
if (!file.exists(local_manifest)) {
|
||||||
cli::cli_abort("Local fixture has no manifest.json at {local_manifest}")
|
cli::cli_abort("Local fixture has no manifest.json at {local_manifest}")
|
||||||
}
|
}
|
||||||
return(jsonlite::fromJSON(local_manifest, simplifyVector = FALSE))
|
return(tryCatch(
|
||||||
|
jsonlite::fromJSON(local_manifest, simplifyVector = FALSE),
|
||||||
|
error = function(e) .abort_invalid_manifest(source = local_manifest, parse_error = e)
|
||||||
|
))
|
||||||
}
|
}
|
||||||
|
|
||||||
cache_path <- file.path(cache_dir, "manifest.json")
|
cache_path <- file.path(cache_dir, "manifest.json")
|
||||||
ttl <- as.integer(.cfg("manifest_ttl_secs"))
|
ttl <- as.integer(.cfg("manifest_ttl_secs"))
|
||||||
|
|
||||||
needs_fetch <- !file.exists(cache_path) ||
|
cache_fresh <- file.exists(cache_path) &&
|
||||||
difftime(Sys.time(), file.info(cache_path)$mtime, units = "secs") > ttl
|
difftime(Sys.time(), file.info(cache_path)$mtime, units = "secs") <= ttl
|
||||||
|
|
||||||
|
# Honor a fresh cache only if its contents still parse as JSON. A previous
|
||||||
|
# version of this package could write HTML directly into the cache; treat
|
||||||
|
# such poisoned caches as if they were missing so the next call recovers.
|
||||||
|
if (cache_fresh) {
|
||||||
|
parsed <- .try_parse_manifest_file(cache_path)
|
||||||
|
if (!is.null(parsed)) return(parsed)
|
||||||
|
}
|
||||||
|
|
||||||
if (needs_fetch) {
|
|
||||||
resp <- httr2::request(paste0(url, "manifest.json")) |>
|
resp <- httr2::request(paste0(url, "manifest.json")) |>
|
||||||
httr2::req_error(is_error = function(r) httr2::resp_status(r) >= 400) |>
|
httr2::req_error(is_error = function(r) httr2::resp_status(r) >= 400) |>
|
||||||
httr2::req_perform()
|
httr2::req_perform()
|
||||||
writeLines(httr2::resp_body_string(resp), cache_path)
|
body <- httr2::resp_body_string(resp)
|
||||||
}
|
|
||||||
|
|
||||||
jsonlite::fromJSON(cache_path, simplifyVector = FALSE)
|
# Parse BEFORE persisting. If the server returned HTML / a login page /
|
||||||
|
# any non-JSON body with a 2xx status, we must not write it to the cache.
|
||||||
|
parsed <- tryCatch(
|
||||||
|
jsonlite::fromJSON(body, simplifyVector = FALSE),
|
||||||
|
error = function(e) {
|
||||||
|
ct <- tryCatch(httr2::resp_content_type(resp), error = function(e2) NA_character_)
|
||||||
|
.abort_invalid_manifest(
|
||||||
|
source = paste0(url, "manifest.json"),
|
||||||
|
content_type = ct,
|
||||||
|
parse_error = e
|
||||||
|
)
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Atomic write: tmp file alongside cache_path (same filesystem -> no EXDEV)
|
||||||
|
# then rename. Ensures a partial write or interrupted process never
|
||||||
|
# replaces a previously-good cache.
|
||||||
|
if (!dir.exists(cache_dir)) dir.create(cache_dir, recursive = TRUE)
|
||||||
|
tmp <- paste0(cache_path, ".tmp.", Sys.getpid())
|
||||||
|
on.exit(if (file.exists(tmp)) unlink(tmp), add = TRUE)
|
||||||
|
writeLines(body, tmp)
|
||||||
|
file.rename(tmp, cache_path)
|
||||||
|
parsed
|
||||||
}
|
}
|
||||||
|
|
||||||
#' @noRd
|
#' @noRd
|
||||||
@@ -36,11 +128,21 @@
|
|||||||
}
|
}
|
||||||
|
|
||||||
#' @noRd
|
#' @noRd
|
||||||
.validate_schema <- function(manifest, expected_version) {
|
#' Schema v6 (FIPS geography harmonization, 2026-07-22) is accepted alongside
|
||||||
if (manifest$schema_version != expected_version) {
|
#' 4/5. v6 renamed the long table's fips_state_code/fips_county_code to
|
||||||
|
#' fips_state_asof/fips_county_asof and added cog_legacy_state/
|
||||||
|
#' cog_legacy_county (26 -> 28 cols); this package references NONE of those
|
||||||
|
#' columns, so no code change was needed. NOTE the SILENT semantic change for
|
||||||
|
#' any consumer of the raw long table: long fips_state/fips_county are now
|
||||||
|
#' PRESENT/harmonized geography (current county identity carried back to every
|
||||||
|
#' year, matching canonical_fips_xwalk) rather than as-of-year; as-of-year
|
||||||
|
#' moved to the *_asof columns. This package's own geography always came from
|
||||||
|
#' the xwalk (already present-based), so behaviour is unchanged.
|
||||||
|
.validate_schema <- function(manifest, supported = c(4L, 5L, 6L, 7L)) {
|
||||||
|
if (!manifest$schema_version %in% supported) {
|
||||||
cli::cli_abort(c(
|
cli::cli_abort(c(
|
||||||
"Corpus schema version mismatch.",
|
"Corpus schema version mismatch.",
|
||||||
x = "Package expects schema_version = {expected_version}; corpus has {manifest$schema_version}.",
|
x = "Package supports schema_version in {paste(supported, collapse = ', ')}; corpus has {manifest$schema_version}.",
|
||||||
i = "Update uscogdata (install.packages or pak::pkg_install) or re-publish corpus."
|
i = "Update uscogdata (install.packages or pak::pkg_install) or re-publish corpus."
|
||||||
))
|
))
|
||||||
}
|
}
|
||||||
@@ -54,3 +156,18 @@
|
|||||||
}
|
}
|
||||||
|
|
||||||
`%||%` <- function(a, b) if (is.null(a) || (length(a) == 1 && is.na(a))) b else a
|
`%||%` <- function(a, b) if (is.null(a) || (length(a) == 1 && is.na(a))) b else a
|
||||||
|
|
||||||
|
#' Return the parsed corpus manifest for the active session.
|
||||||
|
#'
|
||||||
|
#' Opens a session (connecting to the configured corpus) if none is active,
|
||||||
|
#' then returns the manifest exactly as parsed from `manifest.json`. Useful
|
||||||
|
#' for consumers that need the published year range (`years` block, schema
|
||||||
|
#' v5+) or the partition list without issuing a data query.
|
||||||
|
#'
|
||||||
|
#' @return Named list: `schema_version`, `built_at`, `pipeline_commit`,
|
||||||
|
#' `data_vintage`, `scope`, `years` (schema v5+), `schema`, `files`.
|
||||||
|
#' @export
|
||||||
|
cog_manifest <- function() {
|
||||||
|
.ensure_session()
|
||||||
|
.uscogdata_env$manifest
|
||||||
|
}
|
||||||
|
|||||||
@@ -2,32 +2,43 @@
|
|||||||
|
|
||||||
#' Find peer governments by similarity criteria
|
#' Find peer governments by similarity criteria
|
||||||
#'
|
#'
|
||||||
#' Selects peer governments from `canonical_fips_xwalk` by combinations of
|
#' Selects peer governments by combinations of government type, state, and
|
||||||
#' government type, state, and population range. Peers are ordered by
|
#' population range at a chosen `year`. Peers are ordered by `|log(pop_ratio)|`
|
||||||
#' `|log(pop_ratio)|` ascending (closest to the target's population first).
|
#' ascending (closest to the target's population first).
|
||||||
#'
|
#'
|
||||||
#' @param target_govid Character scalar — `canonical_govid` of the target.
|
#' @param target_govid Character scalar — `canonical_govid` of the target.
|
||||||
|
#' @param year Integer scalar. Cohort vintage. When `NULL` (default), uses the
|
||||||
|
#' most recent year for which the target has an observed population in
|
||||||
|
#' `gov_population_yearly`.
|
||||||
#' @param same_type If `TRUE` (default) restrict peers to the target's
|
#' @param same_type If `TRUE` (default) restrict peers to the target's
|
||||||
#' `govs_type`.
|
#' `govs_type`.
|
||||||
#' @param same_state If `TRUE` restrict peers to the target's `fips_state`.
|
#' @param same_state If `TRUE` restrict peers to the target's `fips_state`.
|
||||||
#' Default `FALSE`.
|
#' Default `FALSE`.
|
||||||
#' @param pop_range Length-2 numeric vector giving lower/upper bounds.
|
#' @param pop_range Length-2 numeric vector giving lower/upper bounds.
|
||||||
#' @param is_ratio If `TRUE` (default) `pop_range` is multiplied by the
|
#' @param is_ratio If `TRUE` (default) `pop_range` is multiplied by the
|
||||||
#' target's `population_acs` to produce absolute bounds. If `FALSE`,
|
#' target's population at `year` to produce absolute bounds. If `FALSE`,
|
||||||
#' `pop_range` is interpreted as absolute population counts.
|
#' `pop_range` is interpreted as absolute population counts.
|
||||||
#' @param pop_year Reserved for future use (selecting ACS vintage). Currently
|
|
||||||
#' the corpus has a single snapshot so this argument has no effect.
|
|
||||||
#' @param max_peers Integer cap on the number of peers returned.
|
#' @param max_peers Integer cap on the number of peers returned.
|
||||||
|
#' @param coverage Survey-cycle handling; see [cog_peer_compare()]. Here it
|
||||||
|
#' governs the cohort VINTAGE when `year` is `NULL`: `"census"` snaps to the
|
||||||
|
#' most recent census year with an observed population, so a cohort is not
|
||||||
|
#' built from a sample year in which most of the candidate universe is
|
||||||
|
#' absent. `"consistent"` needs a year range, which cohort selection does not
|
||||||
|
#' have, so it selects like `"all"` and is carried on the result as
|
||||||
|
#' `attr(x, "coverage")` for [cog_peer_compare()].
|
||||||
#' @return Tibble with columns `canonical_govid`, `gov_name`, `fips_state`,
|
#' @return Tibble with columns `canonical_govid`, `gov_name`, `fips_state`,
|
||||||
#' `population_acs`, `pop_ratio`, `rank`.
|
#' `population`, `pop_ratio`, `rank`. The cohort year is attached as
|
||||||
|
#' `attr(x, "cohort_year")`.
|
||||||
#' @export
|
#' @export
|
||||||
cog_find_peers <- function(target_govid,
|
cog_find_peers <- function(target_govid,
|
||||||
|
year = NULL,
|
||||||
same_type = TRUE,
|
same_type = TRUE,
|
||||||
same_state = FALSE,
|
same_state = FALSE,
|
||||||
pop_range = c(0.7, 1.3),
|
pop_range = c(0.7, 1.3),
|
||||||
is_ratio = TRUE,
|
is_ratio = TRUE,
|
||||||
pop_year = NULL,
|
max_peers = 10L,
|
||||||
max_peers = 10L) {
|
coverage = c("all", "census", "consistent")) {
|
||||||
|
coverage <- .validate_coverage(coverage)
|
||||||
if (!is.character(target_govid) || length(target_govid) != 1L) {
|
if (!is.character(target_govid) || length(target_govid) != 1L) {
|
||||||
cli::cli_abort("`target_govid` must be a length-1 character string.")
|
cli::cli_abort("`target_govid` must be a length-1 character string.")
|
||||||
}
|
}
|
||||||
@@ -35,65 +46,127 @@ cog_find_peers <- function(target_govid,
|
|||||||
pop_range[1] >= pop_range[2]) {
|
pop_range[1] >= pop_range[2]) {
|
||||||
cli::cli_abort("`pop_range` must be a length-2 numeric with lo < hi.")
|
cli::cli_abort("`pop_range` must be a length-2 numeric with lo < hi.")
|
||||||
}
|
}
|
||||||
|
if (!is.null(year) &&
|
||||||
|
(!(is.numeric(year) || is.integer(year)) || length(year) != 1L)) {
|
||||||
|
cli::cli_abort("`year` must be NULL or a length-1 integer.")
|
||||||
|
}
|
||||||
|
|
||||||
con <- .ensure_session()
|
con <- .ensure_session()
|
||||||
|
|
||||||
target_sql <- sprintf(
|
# Confirm target exists in the xwalk and pull govs_type / fips_state.
|
||||||
"SELECT canonical_govid, gov_name, govs_type, fips_state, population_acs
|
meta_sql <- sprintf(
|
||||||
|
"SELECT canonical_govid, gov_name, govs_type, fips_state
|
||||||
FROM canonical_fips_xwalk
|
FROM canonical_fips_xwalk
|
||||||
WHERE canonical_govid = %s",
|
WHERE canonical_govid = %s",
|
||||||
.sql_lit_chr(target_govid)
|
.sql_lit_chr(target_govid)
|
||||||
)
|
)
|
||||||
target <- DBI::dbGetQuery(con, target_sql)
|
meta <- DBI::dbGetQuery(con, meta_sql)
|
||||||
if (nrow(target) == 0L) {
|
if (nrow(meta) == 0L) {
|
||||||
cli::cli_abort(c(
|
cli::cli_abort(c(
|
||||||
"govid {target_govid} not found in corpus.",
|
"govid {target_govid} not found in corpus.",
|
||||||
i = "v0.1 covers types 0-3 only (state/county/city/township); see vignette('coverage-scope')."
|
i = "v0.1 covers types 0-3 only (state/county/city/township); see vignette('coverage-scope')."
|
||||||
))
|
))
|
||||||
}
|
}
|
||||||
if (is.na(target$population_acs) || target$population_acs <= 0) {
|
|
||||||
cli::cli_abort("Target {target_govid} has missing or non-positive population; cannot build pop_ratio band.")
|
cohort_year <- .resolve_cohort_year(con, target_govid, year, coverage)
|
||||||
|
|
||||||
|
pop_sql <- sprintf(
|
||||||
|
"SELECT population FROM gov_population_yearly
|
||||||
|
WHERE canonical_govid = %s AND year = %d",
|
||||||
|
.sql_lit_chr(target_govid), as.integer(cohort_year)
|
||||||
|
)
|
||||||
|
target_pop <- DBI::dbGetQuery(con, pop_sql)$population
|
||||||
|
if (length(target_pop) == 0L || is.na(target_pop) || target_pop <= 0) {
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"Target {target_govid} has no observed population in {cohort_year}.",
|
||||||
|
i = "Use a year for which population is observed; see gov_population_yearly."
|
||||||
|
))
|
||||||
}
|
}
|
||||||
|
|
||||||
if (isTRUE(is_ratio)) {
|
if (isTRUE(is_ratio)) {
|
||||||
lo <- target$population_acs * pop_range[1]
|
lo <- target_pop * pop_range[1]
|
||||||
hi <- target$population_acs * pop_range[2]
|
hi <- target_pop * pop_range[2]
|
||||||
} else {
|
} else {
|
||||||
lo <- pop_range[1]; hi <- pop_range[2]
|
lo <- pop_range[1]; hi <- pop_range[2]
|
||||||
}
|
}
|
||||||
|
|
||||||
preds <- c(
|
preds <- c(
|
||||||
sprintf("canonical_govid != %s", .sql_lit_chr(target_govid)),
|
sprintf("p.canonical_govid != %s", .sql_lit_chr(target_govid)),
|
||||||
sprintf("population_acs BETWEEN %.6f AND %.6f", lo, hi)
|
sprintf("p.year = %d", as.integer(cohort_year)),
|
||||||
|
sprintf("p.population BETWEEN %.6f AND %.6f", lo, hi)
|
||||||
)
|
)
|
||||||
if (isTRUE(same_type)) preds <- c(preds, sprintf("govs_type = %d", target$govs_type))
|
if (isTRUE(same_type)) preds <- c(preds, sprintf("x.govs_type = %d", meta$govs_type))
|
||||||
if (isTRUE(same_state)) preds <- c(preds, sprintf("fips_state = %s", .sql_lit_chr(target$fips_state)))
|
if (isTRUE(same_state)) preds <- c(preds, sprintf("x.fips_state = %s", .sql_lit_chr(meta$fips_state)))
|
||||||
|
|
||||||
peers_sql <- sprintf(
|
peers_sql <- sprintf(
|
||||||
"SELECT canonical_govid, gov_name, fips_state, population_acs,
|
"SELECT p.canonical_govid, x.gov_name, x.fips_state, p.population,
|
||||||
population_acs / %.6f AS pop_ratio
|
p.population / %.6f AS pop_ratio
|
||||||
FROM canonical_fips_xwalk
|
FROM gov_population_yearly p
|
||||||
|
JOIN canonical_fips_xwalk x USING (canonical_govid)
|
||||||
WHERE %s
|
WHERE %s
|
||||||
ORDER BY ABS(LN(CAST(population_acs AS DOUBLE) / %.6f))
|
ORDER BY ABS(LN(CAST(p.population AS DOUBLE) / %.6f))
|
||||||
LIMIT %d",
|
LIMIT %d",
|
||||||
target$population_acs,
|
target_pop,
|
||||||
paste(preds, collapse = " AND "),
|
paste(preds, collapse = " AND "),
|
||||||
target$population_acs,
|
target_pop,
|
||||||
as.integer(max_peers)
|
as.integer(max_peers)
|
||||||
)
|
)
|
||||||
peers <- tibble::as_tibble(DBI::dbGetQuery(con, peers_sql))
|
peers <- tibble::as_tibble(DBI::dbGetQuery(con, peers_sql))
|
||||||
if (nrow(peers) > 0L) peers$rank <- seq_len(nrow(peers))
|
peers$rank <- if (nrow(peers) > 0L) seq_len(nrow(peers)) else integer(0)
|
||||||
else peers$rank <- integer(0)
|
attr(peers, "cohort_year") <- as.integer(cohort_year)
|
||||||
|
attr(peers, "pop_range") <- as.numeric(pop_range)
|
||||||
|
attr(peers, "is_ratio") <- isTRUE(is_ratio)
|
||||||
|
attr(peers, "coverage") <- coverage
|
||||||
|
attr(peers, "is_census_year") <- .is_census_year(cohort_year)
|
||||||
peers
|
peers
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# `coverage` picks the cohort vintage when the caller did not name one.
|
||||||
|
# "census" snaps to the most recent CENSUS year with an observed population,
|
||||||
|
# so a cohort is not silently built from a sample year in which most of the
|
||||||
|
# candidate universe is absent. "consistent" is a comparison-time concept --
|
||||||
|
# it needs a year RANGE, which cohort selection does not have -- so it selects
|
||||||
|
# like "all" here and is carried on the result for cog_peer_compare().
|
||||||
|
#' @noRd
|
||||||
|
.resolve_cohort_year <- function(con, target_govid, year,
|
||||||
|
coverage = "all") {
|
||||||
|
if (!is.null(year)) return(as.integer(year))
|
||||||
|
if (identical(coverage, "census")) {
|
||||||
|
sql <- sprintf(
|
||||||
|
"SELECT MAX(year) AS y FROM gov_population_yearly
|
||||||
|
WHERE canonical_govid = %s AND year %% 10 IN (2, 7)",
|
||||||
|
.sql_lit_chr(target_govid)
|
||||||
|
)
|
||||||
|
y <- DBI::dbGetQuery(con, sql)$y
|
||||||
|
if (length(y) > 0L && !is.na(y)) return(as.integer(y))
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"{.code coverage = \"census\"} found no census year with an observed population for {target_govid}.",
|
||||||
|
i = "Pass an explicit {.arg year}, or use {.code coverage = \"all\"}."
|
||||||
|
), class = "uscogdata_no_census_years")
|
||||||
|
}
|
||||||
|
sql <- sprintf(
|
||||||
|
"SELECT MAX(year) AS y FROM gov_population_yearly
|
||||||
|
WHERE canonical_govid = %s",
|
||||||
|
.sql_lit_chr(target_govid)
|
||||||
|
)
|
||||||
|
y <- DBI::dbGetQuery(con, sql)$y
|
||||||
|
if (length(y) == 0L || is.na(y)) {
|
||||||
|
cli::cli_abort(
|
||||||
|
"Target {target_govid} has no observed population in any year."
|
||||||
|
)
|
||||||
|
}
|
||||||
|
as.integer(y)
|
||||||
|
}
|
||||||
|
|
||||||
#' Compare a target government against a peer set
|
#' Compare a target government against a peer set
|
||||||
#'
|
#'
|
||||||
#' Pulls spending for the target plus a peer set (either a
|
#' Pulls spending for the target plus a peer set (either a
|
||||||
#' [cog_find_peers()] result or a character vector of `canonical_govid`) and
|
#' [cog_find_peers()] result or a character vector of `canonical_govid`) and
|
||||||
#' appends peer-distribution summary rows (`summary_p25`, `summary_p50`,
|
#' appends peer-distribution summary rows (`summary_p25`, `summary_p50`,
|
||||||
#' `summary_p75`) so the result can be faceted by `role` in a single ggplot
|
#' `summary_p75`) so the result can be faceted by `role` in a single ggplot
|
||||||
#' call.
|
#' call. Those summary rows are quantiles **within each category**, not
|
||||||
|
#' quantiles of each peer's total — see the `@return` section before summing
|
||||||
|
#' them.
|
||||||
#'
|
#'
|
||||||
#' @param target_govid Character scalar.
|
#' @param target_govid Character scalar.
|
||||||
#' @param peers A tibble from [cog_find_peers()] or a character vector of
|
#' @param peers A tibble from [cog_find_peers()] or a character vector of
|
||||||
@@ -103,18 +176,109 @@ cog_find_peers <- function(target_govid,
|
|||||||
#' @param per_capita Default `TRUE` — peer compare usually normalizes by
|
#' @param per_capita Default `TRUE` — peer compare usually normalizes by
|
||||||
#' population.
|
#' population.
|
||||||
#' @param adjust_to_year Integer base year for CPI-U conversion or `NULL`.
|
#' @param adjust_to_year Integer base year for CPI-U conversion or `NULL`.
|
||||||
|
#' @param expenditure_concept `"primary"` (default), `"direct"`, or
|
||||||
|
#' `"total"` -- see [cog_spending()] for the three concepts. `"total"` is
|
||||||
|
#' refused here because combining Total across peer sets counts
|
||||||
|
#' intergovernmental transfers twice; `"primary"` and `"direct"` combine
|
||||||
|
#' safely.
|
||||||
|
#' @param coverage How to handle the Census of Governments survey cycle,
|
||||||
|
#' which is a **complete census only in years ending in 2 and 7** -- every
|
||||||
|
#' other year is a sample, and the sample varies enormously (on the bundled
|
||||||
|
#' fixture, Wisconsin's 608-city universe reports 597 governments in FY2012
|
||||||
|
#' and 112 in FY2019).
|
||||||
|
#'
|
||||||
|
#' * `"all"` (default) -- every unit that reported that year. Unchanged
|
||||||
|
#' behaviour, so existing code keeps working.
|
||||||
|
#' * `"census"` -- census years only. Aborts if the requested range holds
|
||||||
|
#' none, rather than silently returning nothing.
|
||||||
|
#' * `"consistent"` -- only units reporting in *every* requested year, giving
|
||||||
|
#' a balanced panel.
|
||||||
|
#'
|
||||||
|
#' Regardless of mode, `provenance$coverage` always carries per-year
|
||||||
|
#' `n_units_reporting`, `n_units_expected` and `is_census_year`, and
|
||||||
|
#' `provenance$coverage_mode` records the mode. `is_census_year` is a
|
||||||
|
#' statement about the **survey calendar**, never a claim of completeness:
|
||||||
|
#' FY1967 is a census year in which only 97 of Wisconsin's 608 cities
|
||||||
|
#' report. `n_units_reporting` is the number that tells the truth.
|
||||||
|
#'
|
||||||
|
#' The comparison target is exempt from `"consistent"` balancing -- it is the
|
||||||
|
#' subject of the comparison, not a member of the cohort -- and the
|
||||||
|
#' `summary_*` quantiles are computed AFTER the filter, so they describe the
|
||||||
|
#' cohort actually returned. `n_units_reporting` counts peers only, against
|
||||||
|
#' the cohort size: "3 of your 15 peers reported in FY2019".
|
||||||
#' @return Tibble matching [cog_spending()]'s columns, plus a `role`
|
#' @return Tibble matching [cog_spending()]'s columns, plus a `role`
|
||||||
#' column taking values `"target"`, `"peer"`, `"summary_p25"`,
|
#' column taking values `"target"`, `"peer"`, `"summary_p25"`,
|
||||||
#' `"summary_p50"`, or `"summary_p75"`, and `target_rank` (target's rank
|
#' `"summary_p50"`, or `"summary_p75"`, `target_rank` (target's rank
|
||||||
#' among target+peers at `max(years)`, NA for other rows). Provenance
|
#' among target+peers at `max(years)`, NA for other rows), and
|
||||||
#' attribute reports `verb = "cog_peer_compare"` and `peer_count`.
|
#' `cohort_year` (the year used to build the peer cohort, read from
|
||||||
|
#' `attr(peers, "cohort_year")`; `NA` when `peers` was a bare character
|
||||||
|
#' vector). Provenance reports `verb = "cog_peer_compare"`, `peer_count`,
|
||||||
|
#' `cohort_year`, and `cohort_govids`.
|
||||||
|
#'
|
||||||
|
#' **The `summary_*` rows are per-category quantiles: they are not additive.**
|
||||||
|
#' Each one is computed **within each `(year, spend_subtype,
|
||||||
|
#' category)` cell** across the peer set, so a `summary_p50` row is *the
|
||||||
|
#' median peer's value in that one category*, not *the value of the median
|
||||||
|
#' peer's total*. The median peer for Police and the median peer for Fire
|
||||||
|
#' are usually different governments, so summing `summary_*` rows across
|
||||||
|
#' categories does not give any peer's total and misstates the band it
|
||||||
|
#' appears to describe — measured at −32.7% to +251.0% across 24 years on
|
||||||
|
#' one cohort, with a sign flip at FY2012.
|
||||||
|
#'
|
||||||
|
#' Facet by `role` **and** `category` (the documented use, and what the
|
||||||
|
#' rows are built for). For a genuine "median peer's total spending" line,
|
||||||
|
#' sum each peer's own categories first and take the quantile of those
|
||||||
|
#' per-government totals:
|
||||||
|
#'
|
||||||
|
#' ```r
|
||||||
|
#' library(dplyr)
|
||||||
|
#' cmp |>
|
||||||
|
#' filter(role %in% c("target", "peer")) |>
|
||||||
|
#' group_by(year, role, canonical_govid) |>
|
||||||
|
#' summarise(total = sum(amt_per_capita_real, na.rm = TRUE), .groups = "drop") |>
|
||||||
|
#' filter(role == "peer") |>
|
||||||
|
#' group_by(year) |>
|
||||||
|
#' summarise(p50 = quantile(total, 0.5, na.rm = TRUE))
|
||||||
|
#' ```
|
||||||
|
#' @section Reading `coverage`:
|
||||||
|
#' `provenance$coverage` reports `n_units_reporting` against
|
||||||
|
#' `n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
||||||
|
#' it counts cohort members with rows for the category you asked for, not
|
||||||
|
#' cohort members collected that year.** A government that was surveyed and
|
||||||
|
#' genuinely spends nothing in that category is indistinguishable here from one
|
||||||
|
#' that was never surveyed.
|
||||||
|
#'
|
||||||
|
#' The ratio is therefore **not a response rate** and must not be used as one.
|
||||||
|
#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||||
|
#' `category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||||
|
#' contract policing to the county sheriff, not non-response.
|
||||||
|
#'
|
||||||
|
#' The comparison that *is* valid is the same category across a census year
|
||||||
|
#' (ending in 2 or 7) and a sample year, where the real-zero component is
|
||||||
|
#' roughly constant and the difference reflects the survey cycle. `is_census_year`
|
||||||
|
#' marks which is which.
|
||||||
#' @export
|
#' @export
|
||||||
cog_peer_compare <- function(target_govid, peers, category, years,
|
cog_peer_compare <- function(target_govid, peers, category, years,
|
||||||
per_capita = TRUE, adjust_to_year = NULL) {
|
per_capita = TRUE, adjust_to_year = NULL,
|
||||||
|
expenditure_concept = c("primary", "direct", "total"),
|
||||||
|
coverage = c("all", "census", "consistent")) {
|
||||||
call <- match.call()
|
call <- match.call()
|
||||||
|
expenditure_concept <- match.arg(expenditure_concept)
|
||||||
|
coverage <- .validate_coverage(coverage)
|
||||||
|
if (identical(expenditure_concept, "total")) {
|
||||||
|
.abort_concept_not_aggregatable("cog_peer_compare")
|
||||||
|
}
|
||||||
if (!is.character(target_govid) || length(target_govid) != 1L) {
|
if (!is.character(target_govid) || length(target_govid) != 1L) {
|
||||||
cli::cli_abort("`target_govid` must be a length-1 character string.")
|
cli::cli_abort("`target_govid` must be a length-1 character string.")
|
||||||
}
|
}
|
||||||
|
cohort_year <- if (is.data.frame(peers)) {
|
||||||
|
ay <- attr(peers, "cohort_year")
|
||||||
|
if (is.null(ay)) NA_integer_ else as.integer(ay)
|
||||||
|
} else {
|
||||||
|
NA_integer_
|
||||||
|
}
|
||||||
|
pop_range <- if (is.data.frame(peers)) attr(peers, "pop_range") else NULL
|
||||||
|
is_ratio <- if (is.data.frame(peers)) attr(peers, "is_ratio") else NULL
|
||||||
peer_govids <- if (is.data.frame(peers)) {
|
peer_govids <- if (is.data.frame(peers)) {
|
||||||
as.character(peers$canonical_govid)
|
as.character(peers$canonical_govid)
|
||||||
} else {
|
} else {
|
||||||
@@ -123,24 +287,49 @@ cog_peer_compare <- function(target_govid, peers, category, years,
|
|||||||
peer_govids <- peer_govids[!is.na(peer_govids) & nzchar(peer_govids)]
|
peer_govids <- peer_govids[!is.na(peer_govids) & nzchar(peer_govids)]
|
||||||
all_govids <- unique(c(target_govid, peer_govids))
|
all_govids <- unique(c(target_govid, peer_govids))
|
||||||
|
|
||||||
r <- cog_spending(all_govids, years, category, per_capita, adjust_to_year)
|
years <- .apply_census_years(years, coverage, "cog_peer_compare")
|
||||||
|
|
||||||
|
r <- cog_spending(all_govids, years, category, per_capita, adjust_to_year,
|
||||||
|
expenditure_concept = expenditure_concept)
|
||||||
r$role <- ifelse(r$canonical_govid == target_govid, "target", "peer")
|
r$role <- ifelse(r$canonical_govid == target_govid, "target", "peer")
|
||||||
|
|
||||||
|
# The target is exempt from balancing: it is the subject of the comparison,
|
||||||
|
# not a member of the cohort being balanced, and dropping it would leave a
|
||||||
|
# peer comparison with nothing to compare. Filtering happens BEFORE the
|
||||||
|
# quantiles below, so a "consistent" cohort's summary rows describe that
|
||||||
|
# cohort rather than the unbalanced one.
|
||||||
|
if (identical(coverage, "consistent")) {
|
||||||
|
r <- .filter_consistent(r, years, keep_ids = target_govid)
|
||||||
|
}
|
||||||
|
|
||||||
value_col <- .peer_value_col(per_capita, adjust_to_year)
|
value_col <- .peer_value_col(per_capita, adjust_to_year)
|
||||||
|
|
||||||
summary_rows <- .peer_summary_rows(r, value_col)
|
summary_rows <- .peer_summary_rows(r, value_col)
|
||||||
out <- dplyr::bind_rows(r, summary_rows)
|
out <- dplyr::bind_rows(r, summary_rows)
|
||||||
rank_val <- .peer_target_rank(r, target_govid, years, value_col)
|
rank_val <- .peer_target_rank(r, target_govid, years, value_col)
|
||||||
out$target_rank <- ifelse(out$role == "target", rank_val, NA_integer_)
|
out$target_rank <- ifelse(out$role == "target", rank_val, NA_integer_)
|
||||||
|
out$cohort_year <- cohort_year
|
||||||
|
|
||||||
prov <- attr(r, "provenance") %||% list()
|
prov <- attr(r, "provenance") %||% list()
|
||||||
prov$verb <- "cog_peer_compare"
|
prov$verb <- "cog_peer_compare"
|
||||||
prov$call <- paste(deparse(call), collapse = " ")
|
prov$call <- paste(deparse(call), collapse = " ")
|
||||||
prov$peer_count <- length(peer_govids)
|
prov$peer_count <- length(peer_govids)
|
||||||
|
prov$cohort_year <- cohort_year
|
||||||
|
prov$cohort_govids <- peer_govids
|
||||||
|
prov$pop_range <- pop_range
|
||||||
|
prov$is_ratio <- is_ratio
|
||||||
prov$target <- list(
|
prov$target <- list(
|
||||||
canonical_govid = target_govid,
|
canonical_govid = target_govid,
|
||||||
gov_name = unique(r$gov_name[r$role == "target"])
|
gov_name = unique(r$gov_name[r$role == "target"])
|
||||||
)
|
)
|
||||||
|
# Counted over PEER rows only, against the cohort size: "3 of your 15 peers
|
||||||
|
# reported in FY2019". Including the target would inflate every count by one
|
||||||
|
# and make a cohort that has entirely stopped reporting look non-empty.
|
||||||
|
prov$coverage_mode <- coverage
|
||||||
|
prov$coverage <- .coverage_table(
|
||||||
|
out, years, length(peer_govids),
|
||||||
|
rows = r[r$role == "peer", , drop = FALSE]
|
||||||
|
)
|
||||||
attr(out, "provenance") <- prov
|
attr(out, "provenance") <- prov
|
||||||
out
|
out
|
||||||
}
|
}
|
||||||
|
|||||||
+74
-3
@@ -4,7 +4,15 @@
|
|||||||
#' @noRd
|
#' @noRd
|
||||||
.build_provenance <- function(verb, call, govid, years, category,
|
.build_provenance <- function(verb, call, govid, years, category,
|
||||||
per_capita, adjust_to_year, result, sql,
|
per_capita, adjust_to_year, result, sql,
|
||||||
subtype_col) {
|
subtype_col, basis = NA_character_,
|
||||||
|
basis_note = NA_character_,
|
||||||
|
expenditure_concept = "primary",
|
||||||
|
expenditure_concept_note = NA_character_,
|
||||||
|
expenditure_concept_direct_suppressed = FALSE,
|
||||||
|
revenue_concept = "general",
|
||||||
|
harmonization = NULL, recipe = NULL,
|
||||||
|
suggestions = list(),
|
||||||
|
completion = NULL) {
|
||||||
manifest <- .uscogdata_env$manifest
|
manifest <- .uscogdata_env$manifest
|
||||||
|
|
||||||
codes <- result[["codes_included"]]
|
codes <- result[["codes_included"]]
|
||||||
@@ -29,6 +37,23 @@
|
|||||||
unique(result$gov_name)
|
unique(result$gov_name)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L))
|
||||||
|
con <- .uscogdata_env$con
|
||||||
|
have_con <- !is.null(con) && DBI::dbIsValid(con)
|
||||||
|
break_refs <- if (have_con) {
|
||||||
|
.build_series_break_refs(con, codes_observed, years, schema_version)
|
||||||
|
} else {
|
||||||
|
character(0)
|
||||||
|
}
|
||||||
|
# Corpus-wide caveats travel separately: they qualify the whole result
|
||||||
|
# rather than one series, and they do not depend on codes_observed (see
|
||||||
|
# .build_corpus_break_refs()).
|
||||||
|
corpus_refs <- if (have_con) {
|
||||||
|
.build_corpus_break_refs(con, years, schema_version)
|
||||||
|
} else {
|
||||||
|
character(0)
|
||||||
|
}
|
||||||
|
|
||||||
list(
|
list(
|
||||||
verb = verb,
|
verb = verb,
|
||||||
call = paste(deparse(call), collapse = " "),
|
call = paste(deparse(call), collapse = " "),
|
||||||
@@ -38,6 +63,26 @@
|
|||||||
),
|
),
|
||||||
years = as.integer(years),
|
years = as.integer(years),
|
||||||
category = category,
|
category = category,
|
||||||
|
basis = basis,
|
||||||
|
basis_note = basis_note,
|
||||||
|
expenditure_concept = expenditure_concept,
|
||||||
|
expenditure_concept_note = expenditure_concept_note,
|
||||||
|
# isTRUE() alone would collapse a deliberate NA (all-categories mode,
|
||||||
|
# where suppression detection cannot run -- see .verb_spendrev()) down to
|
||||||
|
# FALSE, turning "we don't know" back into the false claim this field
|
||||||
|
# exists to avoid. Preserve NA; otherwise normalize to a strict logical.
|
||||||
|
expenditure_concept_direct_suppressed = if (isTRUE(is.na(expenditure_concept_direct_suppressed))) {
|
||||||
|
NA
|
||||||
|
} else {
|
||||||
|
isTRUE(expenditure_concept_direct_suppressed)
|
||||||
|
},
|
||||||
|
revenue_concept = revenue_concept,
|
||||||
|
harmonization = harmonization %||% list(
|
||||||
|
applied = FALSE, na_rows_excluded = 0L, na_amount_excluded = 0,
|
||||||
|
note = NA_character_
|
||||||
|
),
|
||||||
|
recipe = recipe,
|
||||||
|
suggestions = suggestions,
|
||||||
scope = list(
|
scope = list(
|
||||||
gov_types_included = as.integer(unlist(manifest$scope$gov_types_included)),
|
gov_types_included = as.integer(unlist(manifest$scope$gov_types_included)),
|
||||||
gov_types_excluded = as.integer(unlist(manifest$scope$gov_types_excluded)),
|
gov_types_excluded = as.integer(unlist(manifest$scope$gov_types_excluded)),
|
||||||
@@ -61,9 +106,27 @@
|
|||||||
per_capita = list(
|
per_capita = list(
|
||||||
applied = isTRUE(per_capita),
|
applied = isTRUE(per_capita),
|
||||||
denominator_source = if (isTRUE(per_capita)) {
|
denominator_source = if (isTRUE(per_capita)) {
|
||||||
"ACS 2018-2022 B01003_001 (population_acs from canonical_fips_xwalk)"
|
"Census F-33 population (per-year, from long.population)"
|
||||||
} else {
|
} else {
|
||||||
NA_character_
|
NA_character_
|
||||||
|
},
|
||||||
|
popyear_range = if (isTRUE(per_capita)) {
|
||||||
|
attr(result, ".popyear_range") %||% integer(0)
|
||||||
|
} else {
|
||||||
|
integer(0)
|
||||||
|
},
|
||||||
|
pop_source_counts = if (isTRUE(per_capita)) {
|
||||||
|
ps <- result[["pop_source"]]
|
||||||
|
if (is.null(ps) || length(ps) == 0L) {
|
||||||
|
list(census_f33 = 0L, unavailable = 0L)
|
||||||
|
} else {
|
||||||
|
list(
|
||||||
|
census_f33 = sum(ps == "census_f33", na.rm = TRUE),
|
||||||
|
unavailable = sum(ps == "unavailable", na.rm = TRUE)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
NULL
|
||||||
}
|
}
|
||||||
),
|
),
|
||||||
inflation = list(
|
inflation = list(
|
||||||
@@ -72,7 +135,15 @@
|
|||||||
index = if (is.null(adjust_to_year)) NA_character_ else "CPI-U (BLS CPIAUCSL annual average, bundled)"
|
index = if (is.null(adjust_to_year)) NA_character_ else "CPI-U (BLS CPIAUCSL annual average, bundled)"
|
||||||
)
|
)
|
||||||
),
|
),
|
||||||
series_break_refs = character(0),
|
series_break_refs = break_refs,
|
||||||
|
corpus_break_refs = corpus_refs,
|
||||||
|
# What `complete = TRUE` filled, and the rule it filled by. Always
|
||||||
|
# present so a consumer can read `completion$applied` without testing
|
||||||
|
# for the key -- an absent block and applied = FALSE would otherwise be
|
||||||
|
# indistinguishable from an older reader version.
|
||||||
|
completion = completion %||% list(
|
||||||
|
applied = FALSE, rows_filled = 0L, absence_means = list()
|
||||||
|
),
|
||||||
manifest = list(
|
manifest = list(
|
||||||
schema_version = as.integer(manifest$schema_version),
|
schema_version = as.integer(manifest$schema_version),
|
||||||
pipeline_commit = manifest$pipeline_commit %||% NA_character_,
|
pipeline_commit = manifest$pipeline_commit %||% NA_character_,
|
||||||
|
|||||||
+167
@@ -0,0 +1,167 @@
|
|||||||
|
# R/recipes.R
|
||||||
|
# Harmonization recipes: multi-code, cross-vintage series built by summing a
|
||||||
|
# fixed set of component item codes with per-component weights and
|
||||||
|
# year/gov-type scoping (see the `harmonization_recipes` view, registered
|
||||||
|
# from data/harmonization_recipes.parquet, schema_version >= 5 only).
|
||||||
|
#
|
||||||
|
# Recipes exist because some cross-vintage series can't be expressed as a
|
||||||
|
# 1:1 harmonized_code mapping (basis = "harmonized"): the wide era (pre-2012)
|
||||||
|
# publishes only a combined aggregate row for these families (e.g.
|
||||||
|
# corrections functions 04+05), while the modern era splits them into leaf
|
||||||
|
# codes. A recipe's generic join sums whichever of its component codes are
|
||||||
|
# present for a given year, so the resulting series is continuous across
|
||||||
|
# that format boundary.
|
||||||
|
|
||||||
|
#' List available harmonization recipes
|
||||||
|
#'
|
||||||
|
#' Recipes are multi-code cross-vintage series (see [cog_spending()]'s
|
||||||
|
#' `recipe` argument) catalogued in the corpus's `harmonization_recipes`
|
||||||
|
#' table. Use this to discover valid `recipe` ids.
|
||||||
|
#'
|
||||||
|
#' @param pattern Optional regex matched case-insensitively against
|
||||||
|
#' `recipe_id` or `label`.
|
||||||
|
#' @return Tibble with columns `recipe_id`, `label`, `n_components`,
|
||||||
|
#' `year_min`, `year_max` (the min/max component year coverage), sorted by
|
||||||
|
#' `recipe_id`.
|
||||||
|
#' @export
|
||||||
|
cog_recipes <- function(pattern = NULL) {
|
||||||
|
if (!is.null(pattern) &&
|
||||||
|
(!is.character(pattern) || length(pattern) != 1L)) {
|
||||||
|
cli::cli_abort("`pattern` must be a length-1 character string or NULL.")
|
||||||
|
}
|
||||||
|
con <- .ensure_session()
|
||||||
|
.require_schema_v5(con, .uscogdata_env$manifest, "cog_recipes()")
|
||||||
|
|
||||||
|
where <- if (is.null(pattern)) {
|
||||||
|
""
|
||||||
|
} else {
|
||||||
|
sprintf(
|
||||||
|
"WHERE regexp_matches(recipe_id, %1$s, 'i') OR regexp_matches(label, %1$s, 'i')",
|
||||||
|
.sql_lit_chr(pattern)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
sql <- paste(
|
||||||
|
"SELECT recipe_id, any_value(label) AS label,
|
||||||
|
COUNT(*) AS n_components,
|
||||||
|
MIN(year_min) AS year_min, MAX(year_max) AS year_max
|
||||||
|
FROM harmonization_recipes",
|
||||||
|
where,
|
||||||
|
"GROUP BY recipe_id
|
||||||
|
ORDER BY recipe_id"
|
||||||
|
)
|
||||||
|
out <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||||
|
out$year_min <- as.integer(out$year_min)
|
||||||
|
out$year_max <- as.integer(out$year_max)
|
||||||
|
out$n_components <- as.integer(out$n_components)
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Abort unless the active corpus has schema_version >= 5.
|
||||||
|
#' @noRd
|
||||||
|
.require_schema_v5 <- function(con, manifest, what) {
|
||||||
|
sv <- suppressWarnings(as.integer(manifest$schema_version %||% 0L))
|
||||||
|
if (sv < 5L) {
|
||||||
|
cli::cli_abort(c(
|
||||||
|
sprintf("%s requires corpus schema_version >= 5.", what),
|
||||||
|
x = "Active corpus has schema_version {sv}.",
|
||||||
|
i = "Point USCOGDATA_URL at a schema_version >= 5 corpus to use harmonization recipes."
|
||||||
|
), class = "uscogdata_schema_unsupported")
|
||||||
|
}
|
||||||
|
invisible(sv)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Abort with the valid id list unless `recipe_id` exists in the catalog.
|
||||||
|
#' @noRd
|
||||||
|
.validate_recipe_id <- function(con, recipe_id) {
|
||||||
|
ids <- DBI::dbGetQuery(
|
||||||
|
con, "SELECT DISTINCT recipe_id FROM harmonization_recipes"
|
||||||
|
)$recipe_id
|
||||||
|
if (!recipe_id %in% ids) {
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"Unknown recipe = {.val {recipe_id}}.",
|
||||||
|
i = "Valid ids: {paste(sort(ids), collapse = ', ')}",
|
||||||
|
i = "See cog_recipes() for labels and year coverage."
|
||||||
|
), class = "uscogdata_unknown_recipe")
|
||||||
|
}
|
||||||
|
invisible(TRUE)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Fetch the component rows for one recipe (label, component codes, scope,
|
||||||
|
#' year ranges, weights) -- both for running the recipe and for the
|
||||||
|
#' `recipe` provenance block.
|
||||||
|
#' @noRd
|
||||||
|
.recipe_components <- function(con, recipe_id) {
|
||||||
|
sql <- sprintf(
|
||||||
|
"SELECT recipe_id, label, component_code, gov_type_scope,
|
||||||
|
year_min, year_max, weight, source_break_ids, notes
|
||||||
|
FROM harmonization_recipes
|
||||||
|
WHERE recipe_id = %s
|
||||||
|
ORDER BY component_code",
|
||||||
|
.sql_lit_chr(recipe_id)
|
||||||
|
)
|
||||||
|
tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Run a recipe's generic join: sum `amt * weight` across whichever
|
||||||
|
#' component codes are present for each (year, canonical_govid), scoped by
|
||||||
|
#' gov_type_scope. Deliberately does NOT filter `NOT is_aggregate`: in the
|
||||||
|
#' wide era (<= 2011) these families' component codes exist ONLY as
|
||||||
|
#' aggregate rows (leaves first appear 2012), so excluding aggregates would
|
||||||
|
#' zero out the wide-era half of every recipe. This is safe by corpus
|
||||||
|
#' construction -- wide-era rows for these codes are aggregate-only, modern
|
||||||
|
#' rows are leaf-only, and every component row is year-scoped via
|
||||||
|
#' `year_min`/`year_max` -- so there is no double-counting. (Checkpoint
|
||||||
|
#' review docs/phase_r_harmonization_review.md § 0.2.)
|
||||||
|
#' @noRd
|
||||||
|
.run_recipe <- function(con, recipe_id, cohort, years) {
|
||||||
|
sql <- sprintf(
|
||||||
|
"SELECT l.year, l.canonical_govid,
|
||||||
|
COALESCE(x.gov_name, l.gov_name) AS gov_name,
|
||||||
|
SUM(l.amt * r.weight) * 1000.0 AS amt_nominal,
|
||||||
|
string_agg(DISTINCT l.item_code, ',' ORDER BY l.item_code) AS codes_included
|
||||||
|
FROM long l
|
||||||
|
JOIN harmonization_recipes r
|
||||||
|
ON l.item_code = r.component_code
|
||||||
|
AND l.year BETWEEN r.year_min AND r.year_max
|
||||||
|
AND (r.gov_type_scope = 'all'
|
||||||
|
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||||
|
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||||
|
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
|
||||||
|
WHERE r.recipe_id = %1$s
|
||||||
|
AND %2$s
|
||||||
|
AND l.year IN (%3$s)
|
||||||
|
GROUP BY 1, 2, 3
|
||||||
|
ORDER BY 1, 2",
|
||||||
|
.sql_lit_chr(recipe_id), .cohort_sql(cohort, "l.canonical_govid"),
|
||||||
|
paste(as.integer(years), collapse = ",")
|
||||||
|
)
|
||||||
|
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||||
|
attr(result, "sql_query") <- sql
|
||||||
|
result
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Shape a raw .run_recipe() result into the standard cog_spending()/
|
||||||
|
#' cog_revenue() column layout: subtype = "recipe", category = the recipe's
|
||||||
|
#' label, aggregate_fallback = FALSE (recipes resolve coverage gaps by
|
||||||
|
#' construction, not by falling back to an aggregate row).
|
||||||
|
#' @noRd
|
||||||
|
.shape_recipe_result <- function(result, subtype_col, label) {
|
||||||
|
sql_query <- attr(result, "sql_query")
|
||||||
|
n <- nrow(result)
|
||||||
|
result[[subtype_col]] <- rep("recipe", n)
|
||||||
|
result$category <- rep(label, n)
|
||||||
|
result$aggregate_fallback <- rep(FALSE, n)
|
||||||
|
result <- result[, c(
|
||||||
|
"year", "canonical_govid", "gov_name", subtype_col, "category",
|
||||||
|
"amt_nominal", "codes_included", "aggregate_fallback"
|
||||||
|
), drop = FALSE]
|
||||||
|
attr(result, "sql_query") <- sql_query
|
||||||
|
result
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Turn a small data.frame into a list-of-lists (one list per row), the
|
||||||
|
#' shape used for the `recipe$components` provenance block.
|
||||||
|
#' @noRd
|
||||||
|
.df_to_row_list <- function(df) {
|
||||||
|
lapply(seq_len(nrow(df)), function(i) as.list(df[i, , drop = FALSE]))
|
||||||
|
}
|
||||||
+59
-5
@@ -8,22 +8,76 @@
|
|||||||
#' multiplies by 1000 and records the conversion in `provenance`).
|
#' multiplies by 1000 and records the conversion in `provenance`).
|
||||||
#'
|
#'
|
||||||
#' @inheritParams cog_spending
|
#' @inheritParams cog_spending
|
||||||
|
#' @param category Character vector of category names (from
|
||||||
|
#' `summary_categories.category`), or `NULL` for all categories broken out
|
||||||
|
#' one row each. The reserved value `"All Categories"` instead returns a
|
||||||
|
#' single summed row per `(year, canonical_govid, subtype)`, covering every
|
||||||
|
#' category inside the requested concept's subtype scope. It cannot be
|
||||||
|
#' combined with other category names, and it is not the same thing as
|
||||||
|
#' `revenue_concept = "total"`: the concept chooses which subtypes are in
|
||||||
|
#' scope, `"All Categories"` chooses whether rows inside that scope are
|
||||||
|
#' broken out or summed. Because the result keeps one row per
|
||||||
|
#' `revenue_subtype`, filtering the returned frame to
|
||||||
|
#' `revenue_subtype == "own_source"` gives an own-source revenue total.
|
||||||
|
#' @param revenue_concept Which of Census's two published revenue concepts to
|
||||||
|
#' return. Concepts are defined as sets of the crosswalk's `revenue_subtype`
|
||||||
|
#' values -- never as item-code first letters, which cannot classify
|
||||||
|
#' correctly (prefix `Y` spans revenue, expenditure and balance codes, and
|
||||||
|
#' prefix `X` does the same):
|
||||||
|
#'
|
||||||
|
#' * `"general"` (default) -- Census General Revenue: `own_source` +
|
||||||
|
#' `federal` + `state` + `local_aid`. The manual defines this concept by
|
||||||
|
#' subtraction (section 4.3: *"General revenue comprises all revenue
|
||||||
|
#' except that classified as liquor store, utility, or insurance trust
|
||||||
|
#' revenue"*), so utility (`A91`-`A94`), liquor store (`A90`) and
|
||||||
|
#' insurance trust revenue are all excluded.
|
||||||
|
#' * `"total"` -- Census Total Revenue: every revenue subtype, i.e.
|
||||||
|
#' `general` plus utility, liquor store, and insurance trust revenue
|
||||||
|
#' (`Y01`/`Y02`/`Y04`/`Y11`/`Y12`/`Y51`/`Y52` and the employee-retirement
|
||||||
|
#' `X01`/`X02`/`X05`/`X08`).
|
||||||
|
#'
|
||||||
|
#' The two are related by Census's own identity, `Total Revenue = General +
|
||||||
|
#' Utility + Liquor Store + Insurance Trust`.
|
||||||
|
#'
|
||||||
|
#' Note that the employee-retirement (`X`) codes stop at FY2016, when those
|
||||||
|
#' systems moved out of the annual finance file into the separate Annual
|
||||||
|
#' Survey of Public Pensions, so a `"total"` series steps down at the
|
||||||
|
#' FY2016/FY2017 seam for reasons that are about collection scope rather
|
||||||
|
#' than revenue (series breaks `SB197`-`SB202`).
|
||||||
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||||
#' `revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
#' `revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
||||||
#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
||||||
#' `codes_included`, `aggregate_fallback`, `notes`.
|
#' optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`,
|
||||||
|
#' and `value_source` when `complete = TRUE`.
|
||||||
#' @export
|
#' @export
|
||||||
cog_revenue <- function(govid, years, category = NULL,
|
cog_revenue <- function(govid = NULL, years, category = NULL,
|
||||||
per_capita = FALSE, adjust_to_year = NULL) {
|
per_capita = FALSE, adjust_to_year = NULL,
|
||||||
|
basis = c("harmonized", "raw"), recipe = NULL,
|
||||||
|
revenue_concept = c("general", "total"),
|
||||||
|
complete = FALSE, limit = NULL, offset = NULL,
|
||||||
|
state = NULL, type = NULL) {
|
||||||
|
# flow_prefixes no longer classifies rows (crosswalk revenue_subtype
|
||||||
|
# membership does -- General Revenue, i.e. everything except
|
||||||
|
# insurance_trust) -- it only scopes the recipe-suggestion machinery to
|
||||||
|
# this verb's recipe families (see R/suggestions.R).
|
||||||
.verb_spendrev(
|
.verb_spendrev(
|
||||||
verb = "cog_revenue",
|
verb = "cog_revenue",
|
||||||
view = "revenue_annotated",
|
view_base = "revenue_annotated",
|
||||||
subtype_col = "revenue_subtype",
|
subtype_col = "revenue_subtype",
|
||||||
|
flow_prefixes = c("T", "A", "U", "B", "C", "D"),
|
||||||
call = match.call(),
|
call = match.call(),
|
||||||
govid = govid,
|
govid = govid,
|
||||||
years = years,
|
years = years,
|
||||||
category = category,
|
category = category,
|
||||||
per_capita = per_capita,
|
per_capita = per_capita,
|
||||||
adjust_to_year = adjust_to_year
|
adjust_to_year = adjust_to_year,
|
||||||
|
basis = basis,
|
||||||
|
recipe = recipe,
|
||||||
|
revenue_concept = revenue_concept,
|
||||||
|
complete = complete,
|
||||||
|
limit = limit,
|
||||||
|
offset = offset,
|
||||||
|
state = state,
|
||||||
|
type = type
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|||||||
+98
-10
@@ -8,28 +8,88 @@
|
|||||||
#' "place portraits" that compare a city to the surrounding county and
|
#' "place portraits" that compare a city to the surrounding county and
|
||||||
#' containing state on one set of axes.
|
#' containing state on one set of axes.
|
||||||
#'
|
#'
|
||||||
|
#' When `per_capita = TRUE`, rows whose government has no observed
|
||||||
|
#' population in that year (`pop_source == "unavailable"`) are dropped from
|
||||||
|
#' the result. The dropped govids are recorded in
|
||||||
|
#' `provenance$rollup$excluded_govids`. This excludes special districts
|
||||||
|
#' (gov type 4) and school districts (gov type 5) from per-capita rollups
|
||||||
|
#' by design — see `vignette('population-denominators')`.
|
||||||
|
#'
|
||||||
#' @param govids Named list with any non-empty subset of elements named
|
#' @param govids Named list with any non-empty subset of elements named
|
||||||
#' `state`, `county`, `city`. Each element is a character vector of
|
#' `state`, `county`, `city`. Each element is a character vector of
|
||||||
#' `canonical_govid` values. At least one layer required.
|
#' `canonical_govid` values. At least one layer required.
|
||||||
#' @param category Single category name or character vector (passed through
|
#' @param category Single category name or character vector (passed through
|
||||||
#' to [cog_spending()]).
|
#' to [cog_spending()]), or the reserved `"All Categories"` for one summed
|
||||||
|
#' row per `(year, canonical_govid, subtype)` covering every category in the
|
||||||
|
#' concept's scope. `"All Categories"` is the efficient way to build a
|
||||||
|
#' geographic total: without it a caller must issue one rollup per category
|
||||||
|
#' and sum the results themselves.
|
||||||
#' @param years Integer vector of years.
|
#' @param years Integer vector of years.
|
||||||
#' @param per_capita If `TRUE`, per-capita uses each layer's own population
|
#' @param per_capita If `TRUE`, per-capita uses each gov's own per-year
|
||||||
#' from `canonical_fips_xwalk.population_acs`.
|
#' population from `gov_population_yearly`. Govs with missing population
|
||||||
|
#' are excluded from the result.
|
||||||
#' @param adjust_to_year Integer base year for CPI-U conversion, or `NULL`.
|
#' @param adjust_to_year Integer base year for CPI-U conversion, or `NULL`.
|
||||||
|
#' @param expenditure_concept `"primary"` (default), `"direct"`, or
|
||||||
|
#' `"total"` -- see [cog_spending()] for the three concepts. `"total"` is
|
||||||
|
#' refused here because combining Total across multiple layers of
|
||||||
|
#' government double-counts intergovernmental transfers (a state's payment
|
||||||
|
#' to a school district is the same dollar the district reports as its own
|
||||||
|
#' Direct spending); `"primary"` and `"direct"` combine safely.
|
||||||
|
#' @param coverage How to handle the Census of Governments survey cycle,
|
||||||
|
#' which is a **complete census only in years ending in 2 and 7** -- every
|
||||||
|
#' other year is a sample, and the sample varies enormously (on the bundled
|
||||||
|
#' fixture, Wisconsin's 608-city universe reports 597 governments in FY2012
|
||||||
|
#' and 112 in FY2019).
|
||||||
|
#'
|
||||||
|
#' * `"all"` (default) -- every unit that reported that year. Unchanged
|
||||||
|
#' behaviour, so existing code keeps working.
|
||||||
|
#' * `"census"` -- census years only. Aborts if the requested range holds
|
||||||
|
#' none, rather than silently returning nothing.
|
||||||
|
#' * `"consistent"` -- only units reporting in *every* requested year, giving
|
||||||
|
#' a balanced panel.
|
||||||
|
#'
|
||||||
|
#' Regardless of mode, `provenance$coverage` always carries per-year
|
||||||
|
#' `n_units_reporting`, `n_units_expected` and `is_census_year`, and
|
||||||
|
#' `provenance$coverage_mode` records the mode. `is_census_year` is a
|
||||||
|
#' statement about the **survey calendar**, never a claim of completeness:
|
||||||
|
#' FY1967 is a census year in which only 97 of Wisconsin's 608 cities
|
||||||
|
#' report. `n_units_reporting` is the number that tells the truth.
|
||||||
#' @return Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`,
|
#' @return Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`,
|
||||||
#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real` /
|
#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real` /
|
||||||
#' `amt_per_capita_nominal` / `amt_per_capita_real`, `codes_included`,
|
#' `amt_per_capita_nominal` / `amt_per_capita_real`, optional `pop_source`,
|
||||||
#' `aggregate_fallback`, `scope_note`, `notes`. Carries a `provenance`
|
#' `codes_included`, `aggregate_fallback`, `scope_note`, `notes`. Carries a
|
||||||
#' attribute with `verb = "cog_geographic_rollup"` and `layers`.
|
#' `provenance` attribute with `verb = "cog_geographic_rollup"`, `layers`,
|
||||||
|
#' and `rollup$included_govids` / `rollup$excluded_govids`.
|
||||||
|
#' @section Reading `coverage`:
|
||||||
|
#' `provenance$coverage` reports `n_units_reporting` against
|
||||||
|
#' `n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
||||||
|
#' it counts governments with rows for the category you asked for, not
|
||||||
|
#' governments collected that year.** A government that was surveyed and
|
||||||
|
#' genuinely spends nothing in that category is indistinguishable here from one
|
||||||
|
#' that was never surveyed.
|
||||||
|
#'
|
||||||
|
#' The ratio is therefore **not a response rate** and must not be used as one.
|
||||||
|
#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||||
|
#' `category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||||
|
#' contract policing to the county sheriff, not non-response.
|
||||||
|
#'
|
||||||
|
#' The comparison that *is* valid is the same category across a census year
|
||||||
|
#' (ending in 2 or 7) and a sample year, where the real-zero component is
|
||||||
|
#' roughly constant and the difference reflects the survey cycle. `is_census_year`
|
||||||
|
#' marks which is which.
|
||||||
#' @export
|
#' @export
|
||||||
cog_geographic_rollup <- function(govids, category, years,
|
cog_geographic_rollup <- function(govids, category, years,
|
||||||
per_capita = FALSE, adjust_to_year = NULL) {
|
per_capita = FALSE, adjust_to_year = NULL,
|
||||||
|
expenditure_concept = c("primary", "direct", "total"),
|
||||||
|
coverage = c("all", "census", "consistent")) {
|
||||||
call <- match.call()
|
call <- match.call()
|
||||||
|
expenditure_concept <- match.arg(expenditure_concept)
|
||||||
|
coverage <- .validate_coverage(coverage)
|
||||||
|
if (identical(expenditure_concept, "total")) {
|
||||||
|
.abort_concept_not_aggregatable("cog_geographic_rollup")
|
||||||
|
}
|
||||||
.validate_rollup_layers(govids)
|
.validate_rollup_layers(govids)
|
||||||
|
|
||||||
# Accept character vector OR a data.frame with canonical_govid per layer,
|
|
||||||
# so cog_gov_search() output can be piped into one of the layer slots.
|
|
||||||
govids <- lapply(govids, .coerce_govid_input, arg = "govids[[layer]]")
|
govids <- lapply(govids, .coerce_govid_input, arg = "govids[[layer]]")
|
||||||
if (any(lengths(govids) == 0L)) {
|
if (any(lengths(govids) == 0L)) {
|
||||||
cli::cli_abort("Each layer in `govids` must be non-empty after coercion.")
|
cli::cli_abort("Each layer in `govids` must be non-empty after coercion.")
|
||||||
@@ -41,16 +101,44 @@ cog_geographic_rollup <- function(govids, category, years,
|
|||||||
layer = rep(layer_names, lengths(govids))
|
layer = rep(layer_names, lengths(govids))
|
||||||
)
|
)
|
||||||
|
|
||||||
r <- cog_spending(all_govids, years, category, per_capita, adjust_to_year)
|
# coverage = "census" drops non-census years BEFORE the query rather than
|
||||||
|
# after: a sample year's rows are not wanted at all, and fetching them only
|
||||||
|
# to discard them would also let them into the coverage table.
|
||||||
|
years <- .apply_census_years(years, coverage, "cog_geographic_rollup")
|
||||||
|
|
||||||
|
r <- cog_spending(all_govids, years, category, per_capita, adjust_to_year,
|
||||||
|
expenditure_concept = expenditure_concept)
|
||||||
r <- dplyr::left_join(r, layer_map, by = "canonical_govid",
|
r <- dplyr::left_join(r, layer_map, by = "canonical_govid",
|
||||||
relationship = "many-to-many")
|
relationship = "many-to-many")
|
||||||
r$scope_note <- .rollup_scope_note(r$layer)
|
r$scope_note <- .rollup_scope_note(r$layer)
|
||||||
|
|
||||||
|
if (identical(coverage, "consistent")) {
|
||||||
|
r <- .filter_consistent(r, years)
|
||||||
|
}
|
||||||
|
|
||||||
|
excluded <- character(0)
|
||||||
|
if (isTRUE(per_capita) && "pop_source" %in% names(r)) {
|
||||||
|
drop <- r$pop_source == "unavailable"
|
||||||
|
excluded <- unique(r$canonical_govid[drop])
|
||||||
|
r <- r[!drop, , drop = FALSE]
|
||||||
|
}
|
||||||
|
included <- unique(r$canonical_govid)
|
||||||
|
|
||||||
r <- .reorder_rollup_cols(r)
|
r <- .reorder_rollup_cols(r)
|
||||||
|
|
||||||
prov <- attr(r, "provenance")
|
prov <- attr(r, "provenance")
|
||||||
prov$verb <- "cog_geographic_rollup"
|
prov$verb <- "cog_geographic_rollup"
|
||||||
prov$call <- paste(deparse(call), collapse = " ")
|
prov$call <- paste(deparse(call), collapse = " ")
|
||||||
prov$layers <- layer_names
|
prov$layers <- layer_names
|
||||||
|
prov$rollup <- list(
|
||||||
|
included_govids = included,
|
||||||
|
excluded_govids = excluded
|
||||||
|
)
|
||||||
|
# n_units_expected is the universe the CALLER named -- the govids passed in
|
||||||
|
# -- not the national universe. That is what makes the ratio meaningful:
|
||||||
|
# "597 of the 608 Wisconsin cities you asked about reported in FY2012".
|
||||||
|
prov$coverage_mode <- coverage
|
||||||
|
prov$coverage <- .coverage_table(r, years, length(unique(all_govids)))
|
||||||
attr(r, "provenance") <- prov
|
attr(r, "provenance") <- prov
|
||||||
|
|
||||||
r
|
r
|
||||||
|
|||||||
+34
-14
@@ -6,8 +6,11 @@
|
|||||||
#' the cross-vintage canonical-government registry. Operates in two modes:
|
#' the cross-vintage canonical-government registry. Operates in two modes:
|
||||||
#'
|
#'
|
||||||
#' * **Utility mode** (single `name`, the original behavior): returns all
|
#' * **Utility mode** (single `name`, the original behavior): returns all
|
||||||
#' rows whose `gov_name` matches the regex case-insensitively, sorted by
|
#' rows whose `gov_name` contains `name` as a **literal, case-insensitive
|
||||||
#' `population_acs` descending. Useful for exploratory lookups.
|
#' substring**, sorted by `population_acs` descending. Useful for
|
||||||
|
#' exploratory lookups. Regex metacharacters in `name` are escaped, so a
|
||||||
|
#' government is findable by its own complete name even when that name
|
||||||
|
#' contains parentheses or a period.
|
||||||
#' * **Basket mode** (`length(name) > 1`): resolves each input row to a
|
#' * **Basket mode** (`length(name) > 1`): resolves each input row to a
|
||||||
#' single canonical govid and returns a tibble in input order, suitable
|
#' single canonical govid and returns a tibble in input order, suitable
|
||||||
#' for piping straight into [cog_spending()] / [cog_revenue()] /
|
#' for piping straight into [cog_spending()] / [cog_revenue()] /
|
||||||
@@ -19,7 +22,8 @@
|
|||||||
#' 1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`.
|
#' 1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`.
|
||||||
#' 2. **Exact pass:** case-insensitive equality against `gov_name`.
|
#' 2. **Exact pass:** case-insensitive equality against `gov_name`.
|
||||||
#' Single hit -> resolved. Multiple -> step 4.
|
#' Single hit -> resolved. Multiple -> step 4.
|
||||||
#' 3. **Substring fallback:** case-insensitive regex against `gov_name`.
|
#' 3. **Substring fallback:** case-insensitive literal substring against
|
||||||
|
#' `gov_name` (metacharacters escaped).
|
||||||
#' Single hit -> resolved (`match_method = "substring"`). Zero hits ->
|
#' Single hit -> resolved (`match_method = "substring"`). Zero hits ->
|
||||||
#' `status = "no_match"`. Multiple hits -> step 4.
|
#' `status = "no_match"`. Multiple hits -> step 4.
|
||||||
#' 4. **Disambiguation:** if matches share one `govs_type`, pick the
|
#' 4. **Disambiguation:** if matches share one `govs_type`, pick the
|
||||||
@@ -48,7 +52,7 @@
|
|||||||
#' [cog_spending()], [cog_revenue()].
|
#' [cog_spending()], [cog_revenue()].
|
||||||
#' @examples
|
#' @examples
|
||||||
#' \dontrun{
|
#' \dontrun{
|
||||||
#' # Utility mode — exploratory regex lookup
|
#' # Utility mode — exploratory substring lookup
|
||||||
#' cog_gov_search("broward", state = "FL")
|
#' cog_gov_search("broward", state = "FL")
|
||||||
#'
|
#'
|
||||||
#' # Basket mode — resolve a known cohort
|
#' # Basket mode — resolve a known cohort
|
||||||
@@ -98,9 +102,16 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) {
|
|||||||
if (!is.character(name) || length(name) != 1L) {
|
if (!is.character(name) || length(name) != 1L) {
|
||||||
cli::cli_abort("`name` must be a length-1 character string.")
|
cli::cli_abort("`name` must be a length-1 character string.")
|
||||||
}
|
}
|
||||||
|
# Escaped, so `name` is a literal case-insensitive substring -- the same
|
||||||
|
# treatment basket mode has always given it. Interpolating it raw made a
|
||||||
|
# government unfindable by its own name whenever that name contains a
|
||||||
|
# metacharacter (FREDONIA (BRISCOE) CITY), turned a bare "." into a
|
||||||
|
# match-everything wildcard, and let malformed pattern text reach the
|
||||||
|
# engine as an error -- which cog-api surfaced as a 500, reachable by
|
||||||
|
# typing a real name one character at a time (uscogdata#16, F-025).
|
||||||
preds <- c(preds,
|
preds <- c(preds,
|
||||||
sprintf("regexp_matches(gov_name, %s, 'i')",
|
sprintf("regexp_matches(gov_name, %s, 'i')",
|
||||||
.sql_lit_chr(name)))
|
.sql_lit_chr(.escape_regex(name))))
|
||||||
}
|
}
|
||||||
if (!is.null(state)) {
|
if (!is.null(state)) {
|
||||||
st_fips <- .coerce_state_to_fips(state)
|
st_fips <- .coerce_state_to_fips(state)
|
||||||
@@ -126,17 +137,19 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) {
|
|||||||
canonical_govid = character(0), gov_name = character(0),
|
canonical_govid = character(0), gov_name = character(0),
|
||||||
govs_type = integer(0), type_label = character(0),
|
govs_type = integer(0), type_label = character(0),
|
||||||
fips_state = character(0), fips_county = character(0),
|
fips_state = character(0), fips_county = character(0),
|
||||||
fips_place = character(0), first_year = integer(0),
|
fips_place = character(0), legacy_govs_id = character(0),
|
||||||
last_year = integer(0), population_acs = integer(0),
|
first_year = integer(0), last_year = integer(0),
|
||||||
confidence = character(0)
|
census_geoid = character(0), population_acs = integer(0),
|
||||||
|
pop_confidence = character(0), id_source = character(0)
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
#' @noRd
|
#' @noRd
|
||||||
.escape_regex <- function(x) {
|
.escape_regex <- function(x) {
|
||||||
# Backslash-escape POSIX regex metacharacters so `name` is treated as a
|
# Backslash-escape POSIX regex metacharacters so `name` is treated as a
|
||||||
# literal substring in the DuckDB regexp_matches call (substring fallback
|
# literal substring in the DuckDB regexp_matches call. Used by BOTH modes:
|
||||||
# only; utility-mode intentionally preserves regex behavior).
|
# utility mode used to interpolate raw, which was a defect rather than a
|
||||||
|
# feature -- see the call site and uscogdata#16.
|
||||||
gsub("([\\^$.|?*+(){}\\[\\]])", "\\\\\\1", x, perl = TRUE)
|
gsub("([\\^$.|?*+(){}\\[\\]])", "\\\\\\1", x, perl = TRUE)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -171,11 +184,18 @@ cog_gov_search <- function(name = NULL, state = NULL, type = NULL) {
|
|||||||
if (!is.character(state) || length(state) != 1L) {
|
if (!is.character(state) || length(state) != 1L) {
|
||||||
cli::cli_abort("`state` must be a 2-letter USPS abbrev or a FIPS integer.")
|
cli::cli_abort("`state` must be a 2-letter USPS abbrev or a FIPS integer.")
|
||||||
}
|
}
|
||||||
fips <- .state_abbrev_to_fips[[toupper(state)]]
|
# Membership tested before the lookup, not after: `.state_abbrev_to_fips` is
|
||||||
if (is.null(fips)) {
|
# a named CHARACTER vector, and `[[` on a name it does not carry throws
|
||||||
cli::cli_abort("Unknown state abbreviation: {state}.")
|
# base R's "subscript out of bounds" rather than returning NULL -- which
|
||||||
|
# made the curated message below unreachable dead code. Reported as a bare
|
||||||
|
# subscript error, `cog_gov_search(state = "ZZ")` gave no hint that the
|
||||||
|
# argument wants a postal abbreviation.
|
||||||
|
key <- toupper(state)
|
||||||
|
if (!key %in% names(.state_abbrev_to_fips)) {
|
||||||
|
cli::cli_abort("Unknown state abbreviation: {state}.",
|
||||||
|
class = "uscogdata_unknown_state")
|
||||||
}
|
}
|
||||||
fips
|
.state_abbrev_to_fips[[key]]
|
||||||
}
|
}
|
||||||
|
|
||||||
# USPS state / territory abbreviation -> 2-digit FIPS code.
|
# USPS state / territory abbreviation -> 2-digit FIPS code.
|
||||||
|
|||||||
@@ -0,0 +1,54 @@
|
|||||||
|
# R/series_breaks.R
|
||||||
|
# Populates prov$series_break_refs (schema in inst/schemas/provenance-v1.json
|
||||||
|
# defines the field; it was always present but always empty pre-Phase-R2)
|
||||||
|
# with the ids of any catalogued series break whose fin_code appears among
|
||||||
|
# the result's observed item codes and whose break_year falls inside the
|
||||||
|
# requested year span -- the "break warnings in the provenance envelope"
|
||||||
|
# spec § 5 promises downstream consumers (cog-api passes provenance through
|
||||||
|
# verbatim). schema_version >= 5 only: series_breaks_pq isn't registered on
|
||||||
|
# an older corpus.
|
||||||
|
|
||||||
|
#' @noRd
|
||||||
|
.build_series_break_refs <- function(con, codes_observed, years, schema_version) {
|
||||||
|
if (schema_version < 5L || length(codes_observed) == 0L) return(character(0))
|
||||||
|
sql <- sprintf(
|
||||||
|
"SELECT DISTINCT break_id
|
||||||
|
FROM series_breaks_pq
|
||||||
|
WHERE fin_code IN (%s) AND fin_code <> 'ALL'
|
||||||
|
AND break_year BETWEEN %d AND %d
|
||||||
|
ORDER BY break_id",
|
||||||
|
.sql_lit_chr(codes_observed), min(as.integer(years)), max(as.integer(years))
|
||||||
|
)
|
||||||
|
DBI::dbGetQuery(con, sql)$break_id
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Corpus-wide caveats: catalogued breaks whose `fin_code` is the literal
|
||||||
|
#' `"ALL"` rather than an item code. They qualify the whole result, so they
|
||||||
|
#' cannot be matched the way `.build_series_break_refs()` matches -- no row's
|
||||||
|
#' `item_code` is ever `"ALL"`, which is exactly why they reached no user
|
||||||
|
#' before uscogdata#19. Selection is on the break_year window alone: which
|
||||||
|
#' codes a result happens to contain is irrelevant to a caveat about the
|
||||||
|
#' corpus.
|
||||||
|
#'
|
||||||
|
#' All four catalogued entries are *boundary* caveats (dollar precision
|
||||||
|
#' across 1976/1977, imputation exclusion from 2002, the dense -> sparse
|
||||||
|
#' representation change at 2012, the id scheme change at 2017), so the same
|
||||||
|
#' `break_year BETWEEN min(years) AND max(years)` rule the code-specific
|
||||||
|
#' path uses is the right one -- a request that never crosses the boundary
|
||||||
|
#' is not affected by it.
|
||||||
|
#'
|
||||||
|
#' Returned separately from `series_break_refs` so a consumer can tell a
|
||||||
|
#' whole-result caveat from a break in one series; the two are disjoint by
|
||||||
|
#' construction.
|
||||||
|
#' @noRd
|
||||||
|
.build_corpus_break_refs <- function(con, years, schema_version) {
|
||||||
|
if (schema_version < 5L || length(years) == 0L) return(character(0))
|
||||||
|
sql <- sprintf(
|
||||||
|
"SELECT DISTINCT break_id
|
||||||
|
FROM series_breaks_pq
|
||||||
|
WHERE fin_code = 'ALL' AND break_year BETWEEN %d AND %d
|
||||||
|
ORDER BY break_id",
|
||||||
|
min(as.integer(years)), max(as.integer(years))
|
||||||
|
)
|
||||||
|
DBI::dbGetQuery(con, sql)$break_id
|
||||||
|
}
|
||||||
+36
-2
@@ -2,16 +2,27 @@
|
|||||||
|
|
||||||
#' Internal: open session, register views, cache manifest.
|
#' Internal: open session, register views, cache manifest.
|
||||||
#' Not exported. Called lazily by verbs via .ensure_session().
|
#' Not exported. Called lazily by verbs via .ensure_session().
|
||||||
|
#'
|
||||||
|
#' `threads` and `memory_limit` default to the resolved configuration and are
|
||||||
|
#' applied as pragmas on the new connection. When both resolve to NULL -- which
|
||||||
|
#' is the case unless the operator sets one -- NO pragma is issued at all, so an
|
||||||
|
#' unconfigured session connects exactly as it did before this argument existed.
|
||||||
#' @noRd
|
#' @noRd
|
||||||
cog_open <- function(url = .resolve_url(),
|
cog_open <- function(url = .resolve_url(),
|
||||||
cache_dir = .resolve_cache_dir()) {
|
cache_dir = .resolve_cache_dir(),
|
||||||
|
threads = .resolve_duckdb_threads(),
|
||||||
|
memory_limit = .resolve_duckdb_memory_limit()) {
|
||||||
|
.check_url_configured(url)
|
||||||
if (!dir.exists(cache_dir)) dir.create(cache_dir, recursive = TRUE)
|
if (!dir.exists(cache_dir)) dir.create(cache_dir, recursive = TRUE)
|
||||||
|
|
||||||
con <- DBI::dbConnect(duckdb::duckdb())
|
con <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
# Before anything else touches the connection: httpfs reads the corpus, and
|
||||||
|
# a remote read should already be bound by whatever budget the operator set.
|
||||||
|
.apply_duckdb_limits(con, threads, memory_limit)
|
||||||
DBI::dbExecute(con, "INSTALL httpfs; LOAD httpfs;")
|
DBI::dbExecute(con, "INSTALL httpfs; LOAD httpfs;")
|
||||||
|
|
||||||
manifest <- .fetch_or_cache_manifest(url, cache_dir)
|
manifest <- .fetch_or_cache_manifest(url, cache_dir)
|
||||||
.validate_schema(manifest, expected_version = 3L)
|
.validate_schema(manifest, supported = c(4L, 5L, 6L, 7L))
|
||||||
.validate_scope(manifest)
|
.validate_scope(manifest)
|
||||||
|
|
||||||
.register_views(con, url, manifest)
|
.register_views(con, url, manifest)
|
||||||
@@ -24,6 +35,26 @@ cog_open <- function(url = .resolve_url(),
|
|||||||
invisible(con)
|
invisible(con)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#' Apply the operator's DuckDB resource budget to a fresh connection.
|
||||||
|
#'
|
||||||
|
#' Split out from cog_open() so the "unset changes nothing" property is one
|
||||||
|
#' readable branch rather than two conditionals buried in the connection path.
|
||||||
|
#' Both settings are session-scoped in DuckDB, so this must run per connection;
|
||||||
|
#' cog_close() discards the connection and the next cog_open() re-resolves,
|
||||||
|
#' which is what makes a changed option take effect on the next session.
|
||||||
|
#' @noRd
|
||||||
|
.apply_duckdb_limits <- function(con, threads, memory_limit) {
|
||||||
|
if (!is.null(threads)) {
|
||||||
|
DBI::dbExecute(con, sprintf("SET threads TO %d", threads))
|
||||||
|
}
|
||||||
|
if (!is.null(memory_limit)) {
|
||||||
|
# Quoted as a string literal: DuckDB's memory_limit takes '4GB', not 4GB.
|
||||||
|
DBI::dbExecute(con, sprintf("SET memory_limit TO %s",
|
||||||
|
.sql_lit_chr(memory_limit)))
|
||||||
|
}
|
||||||
|
invisible(con)
|
||||||
|
}
|
||||||
|
|
||||||
#' @noRd
|
#' @noRd
|
||||||
.ensure_session <- function() {
|
.ensure_session <- function() {
|
||||||
if (is.null(.uscogdata_env$con) ||
|
if (is.null(.uscogdata_env$con) ||
|
||||||
@@ -94,4 +125,7 @@ cog_close <- function() {
|
|||||||
}
|
}
|
||||||
.uscogdata_env$con <- NULL
|
.uscogdata_env$con <- NULL
|
||||||
.uscogdata_env$manifest <- NULL
|
.uscogdata_env$manifest <- NULL
|
||||||
|
.uscogdata_env$balance_caveats_shown <- NULL
|
||||||
|
# Memoised corpus-constant; a different corpus may be mounted next.
|
||||||
|
.uscogdata_env$balance_coverage_windows <- NULL
|
||||||
}
|
}
|
||||||
|
|||||||
+964
-45
File diff suppressed because it is too large
Load Diff
+360
@@ -0,0 +1,360 @@
|
|||||||
|
# R/suggestions.R
|
||||||
|
# Recipe-component-driven signposting. When a basis = "harmonized" query for
|
||||||
|
# a category comes back incomplete in some requested year -- and a
|
||||||
|
# harmonization recipe would actually fill it for this government -- surface
|
||||||
|
# that recipe as a suggestion. "Incomplete" has two forms, and a recipe
|
||||||
|
# qualifies on either:
|
||||||
|
# 1. empty_year -- the result has no rows at all in that year.
|
||||||
|
# 2. suppressed_component -- the result HAS rows, but a component code
|
||||||
|
# carries dollars the verb's own long view structurally excludes
|
||||||
|
# (aggregate-published, or absent from summary_categories). This is
|
||||||
|
# uscogdata#9: Public Welfare kept returning E74/E79 rows while dropping
|
||||||
|
# aggregate-only E67/E68, so form 1 never fired and the caller got a
|
||||||
|
# number a third too low with no signpost at all.
|
||||||
|
#
|
||||||
|
# This is deliberately keyed off the recipe catalog's component codes, not
|
||||||
|
# off harmonization_map rows: no live map row carries a non-blank
|
||||||
|
# suggested_recipe_id (the corpus's wide era exposes split families like
|
||||||
|
# corrections functions 04+05 ONLY as aggregate rows, which basis =
|
||||||
|
# "harmonized" excludes by construction -- there's no NA ruling to hang a
|
||||||
|
# suggestion off of, just a leaf-code absence a recipe happens to fill).
|
||||||
|
# See docs/phase_r_harmonization_review.md § 0.3.
|
||||||
|
#
|
||||||
|
# Scope is deliberately narrow: signposting only runs when the caller
|
||||||
|
# supplied a `category` (an un-scoped, all-categories query has no single
|
||||||
|
# coverage question to answer) and only flags a recipe when the ACTUAL
|
||||||
|
# result has zero rows in a requested year AND the candidate recipe's own
|
||||||
|
# generic join (same join .run_recipe() uses, including its wide-era
|
||||||
|
# aggregate rows) produces at least one row for this government in that
|
||||||
|
# year. Checking presence per-government (not corpus-wide) avoids false
|
||||||
|
# positives from ordinary reporting variance -- most governments don't use
|
||||||
|
# every sibling code in a multi-code category every year, and that is not
|
||||||
|
# a format-boundary gap worth signposting.
|
||||||
|
#
|
||||||
|
# C1(a): for expenditure_concept = "total" callers, `result` here must
|
||||||
|
# already be the Direct-leg subset (the caller filters out
|
||||||
|
# spend_subtype == "intergovernmental" rows before calling in). A gap year
|
||||||
|
# is "the requested year has no Direct rows", never "no rows at all" --
|
||||||
|
# an IG row surviving on a legacy aggregate that Direct excludes must not
|
||||||
|
# read as coverage and cancel the very suggestion that would recover it.
|
||||||
|
|
||||||
|
#' Build the `prov$suggestions` list for a (non-recipe) basis = "harmonized"
|
||||||
|
#' verb call: recipes whose generic join would fill a real gap in `result`.
|
||||||
|
#'
|
||||||
|
#' @param con Active DuckDB connection.
|
||||||
|
#' @param cohort The verb's cohort object (see `.make_cohort()`), naming the
|
||||||
|
#' governments by id, by state/type predicate, or both.
|
||||||
|
#' @param years Integer vector of requested years.
|
||||||
|
#' @param category `category` argument as passed to the verb (character
|
||||||
|
#' vector or `NULL`; suggestions are only computed when non-NULL).
|
||||||
|
#' @param result The verb's already-computed result tibble (post basis
|
||||||
|
#' query, pre per_capita/adjust_to_year), pre-filtered to the Direct leg
|
||||||
|
#' only when the caller's `expenditure_concept = "total"` (see C1(a)).
|
||||||
|
#' @param basis The *resolved* basis (`"harmonized"` or `"raw"`).
|
||||||
|
#' @param flow_prefixes The calling verb's own flow-type prefixes (e.g.
|
||||||
|
#' `c("E", "F", "G")` for `cog_spending()`, `c("T", "A", "U", "B", "C",
|
||||||
|
#' "D")` for `cog_revenue()` -- see `.verb_spendrev()`). Passed through to
|
||||||
|
#' `.attach_ig_counterparts()` to keep the intergovernmental-counterpart
|
||||||
|
#' lookup scoped to the calling verb's own flow family.
|
||||||
|
#' @param long_view Name of the verb's own long view (from
|
||||||
|
#' `.select_long_view()`), passed through to `.suppressed_components()` to
|
||||||
|
#' measure the second qualifying path (uscogdata#9).
|
||||||
|
#' @param all_categories `TRUE` when the caller's `category` is the reserved
|
||||||
|
#' pseudo-category (`.ALL_CATEGORIES`). Defaults to `FALSE` so no other
|
||||||
|
#' caller's behaviour changes. When `TRUE`, the candidate-recipe sub-select
|
||||||
|
#' is scoped by `subtype_col`/`subtype_scope` instead of by `category` --
|
||||||
|
#' symmetric with `.build_verb_sql()`'s own all-categories branch (see
|
||||||
|
#' R/spending.R): the concept's subtype allowlist is the real scope
|
||||||
|
#' boundary, not any literal category value, and
|
||||||
|
#' `.ALL_CATEGORIES` ("All Categories") is never itself a row in
|
||||||
|
#' `summary_categories.category`, so leaving the category-keyed sub-select
|
||||||
|
#' in place here always returned zero candidates and silently disabled
|
||||||
|
#' signposting in all-categories mode (final whole-branch review, finding
|
||||||
|
#' 6).
|
||||||
|
#' @param subtype_col Name of the `summary_categories` subtype column to
|
||||||
|
#' scope by when `all_categories = TRUE` (`"spend_subtype"` or
|
||||||
|
#' `"revenue_subtype"` -- the same value `.build_verb_sql()` already
|
||||||
|
#' receives as its own `subtype_col`). Ignored when `all_categories =
|
||||||
|
#' FALSE`. `NULL` by default.
|
||||||
|
#' @param subtype_scope Character vector of subtype values to scope by when
|
||||||
|
#' `all_categories = TRUE` (the same value `.build_verb_sql()` already
|
||||||
|
#' receives as its own `subtype_scope` -- the concept's subtype allowlist,
|
||||||
|
#' e.g. `.expenditure_concept_subtypes(expenditure_concept)`). Ignored when
|
||||||
|
#' `all_categories = FALSE`. `NULL` by default.
|
||||||
|
#' @return List of `list(recipe_id, label, available_years, hint,
|
||||||
|
#' ig_recipe_id, trigger, suppressed_amount, suppressed_years,
|
||||||
|
#' suppressed_codes)`, possibly empty.
|
||||||
|
#' @noRd
|
||||||
|
.build_suggestions <- function(con, cohort, years, category, result, basis,
|
||||||
|
flow_prefixes, long_view,
|
||||||
|
all_categories = FALSE,
|
||||||
|
subtype_col = NULL, subtype_scope = NULL) {
|
||||||
|
if (!identical(basis, "harmonized") || is.null(category)) return(list())
|
||||||
|
|
||||||
|
# Exclude any recipe that is ITSELF an intergovernmental (M/L) recipe --
|
||||||
|
# i.e. every one of its own component codes is M/L-prefixed. Without this,
|
||||||
|
# a category whose summary_categories rows span both a Direct family
|
||||||
|
# (e.g. E04/E05, "Corrections") and its M/L counterpart (M04/M05, same
|
||||||
|
# category since Task 1) makes the M/L recipe itself (e.g.
|
||||||
|
# `corrections_ig_local_combined`) a raw top-level candidate for a plain
|
||||||
|
# (Direct) cog_spending() call -- following that hint would silently
|
||||||
|
# return intergovernmental dollars under `expenditure_concept = "direct"`
|
||||||
|
# provenance. This is a stronger, unconditional exclusion than the
|
||||||
|
# flow-prefix gate below/in `.attach_ig_counterparts()`: an M/L recipe
|
||||||
|
# should never be suggested as a coverage-gap filler for EITHER verb, not
|
||||||
|
# just kept from being named as the *counterpart* of another suggestion.
|
||||||
|
#
|
||||||
|
# The inner sub-select is the concept boundary (finding 6, final
|
||||||
|
# whole-branch review): in all-categories mode it is scoped by
|
||||||
|
# `subtype_col`/`subtype_scope` -- the same allowlist `.build_verb_sql()`
|
||||||
|
# applies as a WHERE predicate to make the summed result a *concept*, not
|
||||||
|
# by `category` (`.ALL_CATEGORIES` is never a row in
|
||||||
|
# `summary_categories.category`, so a category-keyed sub-select always
|
||||||
|
# came back empty here). The M/L exclusion below is unchanged either way.
|
||||||
|
candidate_scope_sql <- if (isTRUE(all_categories)) {
|
||||||
|
sprintf(
|
||||||
|
"SELECT DISTINCT item_code FROM summary_categories WHERE %s IN (%s)",
|
||||||
|
subtype_col, .sql_lit_chr(subtype_scope)
|
||||||
|
)
|
||||||
|
} else {
|
||||||
|
sprintf(
|
||||||
|
"SELECT DISTINCT item_code FROM summary_categories WHERE category IN (%s)",
|
||||||
|
.sql_lit_chr(category)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
candidates <- DBI::dbGetQuery(con, sprintf(
|
||||||
|
"SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||||
|
WHERE component_code IN (
|
||||||
|
%s
|
||||||
|
)
|
||||||
|
AND recipe_id NOT IN (
|
||||||
|
SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||||
|
WHERE LEFT(component_code, 1) IN ('M', 'L')
|
||||||
|
)",
|
||||||
|
candidate_scope_sql
|
||||||
|
))$recipe_id
|
||||||
|
if (length(candidates) == 0L) return(list())
|
||||||
|
|
||||||
|
result_years <- if (is.null(result) || nrow(result) == 0L) {
|
||||||
|
integer(0)
|
||||||
|
} else {
|
||||||
|
unique(as.integer(result$year))
|
||||||
|
}
|
||||||
|
gap_years <- setdiff(as.integer(years), result_years)
|
||||||
|
|
||||||
|
# Path 2 (uscogdata#9): component dollars this government holds that the
|
||||||
|
# verb's own view structurally excludes. Measured across ALL requested
|
||||||
|
# years, not just gap years -- the whole point is that a year with rows can
|
||||||
|
# still be missing dollars. Scoped to the calling verb's own flow_prefixes
|
||||||
|
# (I1) -- see `.suppressed_components()`'s own roxygen for why.
|
||||||
|
#
|
||||||
|
# This runs unconditionally whenever there are candidates -- an earlier
|
||||||
|
# revision of this fix wave tried a free, in-memory pre-check
|
||||||
|
# (`.needs_suppression_query()`) to skip the round trip on an already-
|
||||||
|
# covered path, but a scoped re-review measured it against the fixture and
|
||||||
|
# found it didn't pay for itself (it skipped ~3% of healthy calls, ~0% of
|
||||||
|
# the multi-govid batch shape it was meant to help, at a net cost increase
|
||||||
|
# once its own always-run metadata query was counted) while adding an
|
||||||
|
# untested exactness invariant -- that `result$codes_included` and this
|
||||||
|
# anti-join share the harmonized `item_code` space -- whose silent
|
||||||
|
# violation would kill signposting, the exact failure class uscogdata#9
|
||||||
|
# exists to prevent. Owner's call: keep this simple; a batch-aware
|
||||||
|
# optimization, if one is worth building, is a separate issue.
|
||||||
|
supp <- .suppressed_components(con, candidates, cohort, years, long_view, flow_prefixes)
|
||||||
|
|
||||||
|
if (length(gap_years) == 0L && nrow(supp) == 0L) return(list())
|
||||||
|
|
||||||
|
meta <- tibble::as_tibble(DBI::dbGetQuery(con, sprintf(
|
||||||
|
"SELECT recipe_id, any_value(label) AS label,
|
||||||
|
MIN(year_min) AS year_min, MAX(year_max) AS year_max
|
||||||
|
FROM harmonization_recipes
|
||||||
|
WHERE recipe_id IN (%s)
|
||||||
|
GROUP BY recipe_id",
|
||||||
|
.sql_lit_chr(candidates)
|
||||||
|
)))
|
||||||
|
|
||||||
|
# Path 1 (unchanged): (recipe, year) pairs the recipe's own generic join
|
||||||
|
# covers for this government, restricted to the gap years.
|
||||||
|
covered <- if (length(gap_years) == 0L) {
|
||||||
|
data.frame(recipe_id = character(0), year = integer(0))
|
||||||
|
} else {
|
||||||
|
DBI::dbGetQuery(con, sprintf(
|
||||||
|
"SELECT DISTINCT r.recipe_id, l.year
|
||||||
|
FROM long l
|
||||||
|
JOIN harmonization_recipes r
|
||||||
|
ON l.item_code = r.component_code
|
||||||
|
AND l.year BETWEEN r.year_min AND r.year_max
|
||||||
|
AND (r.gov_type_scope = 'all'
|
||||||
|
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||||
|
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||||
|
WHERE r.recipe_id IN (%s)
|
||||||
|
AND %s
|
||||||
|
AND l.year IN (%s)",
|
||||||
|
.sql_lit_chr(candidates), .cohort_sql(cohort, "l.canonical_govid"),
|
||||||
|
paste(gap_years, collapse = ",")
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
|
suggestions <- list()
|
||||||
|
for (rid in candidates) {
|
||||||
|
empty_hit <- rid %in% covered$recipe_id
|
||||||
|
s_rows <- supp[supp$recipe_id == rid, , drop = FALSE]
|
||||||
|
supp_hit <- nrow(s_rows) > 0L
|
||||||
|
if (!empty_hit && !supp_hit) next
|
||||||
|
m <- meta[meta$recipe_id == rid, ]
|
||||||
|
suggestions[[length(suggestions) + 1L]] <- list(
|
||||||
|
recipe_id = rid,
|
||||||
|
label = m$label[[1]],
|
||||||
|
available_years = c(as.integer(m$year_min), as.integer(m$year_max)),
|
||||||
|
hint = sprintf("re-run with recipe = '%s'", rid),
|
||||||
|
# An empty year is the stronger claim -- the category returned nothing
|
||||||
|
# at all -- so it wins when both paths qualify. The suppressed_* fields
|
||||||
|
# are still populated, so an empty_year fire also reports its dollars.
|
||||||
|
trigger = if (empty_hit) "empty_year" else "suppressed_component",
|
||||||
|
suppressed_amount = if (supp_hit) sum(s_rows$suppressed_amount) else 0,
|
||||||
|
suppressed_years = if (supp_hit) {
|
||||||
|
sort(unique(as.integer(s_rows$year)))
|
||||||
|
} else {
|
||||||
|
integer(0)
|
||||||
|
},
|
||||||
|
suppressed_codes = if (supp_hit) {
|
||||||
|
sort(unique(unlist(strsplit(s_rows$suppressed_codes, ",", fixed = TRUE))))
|
||||||
|
} else {
|
||||||
|
character(0)
|
||||||
|
}
|
||||||
|
)
|
||||||
|
}
|
||||||
|
.attach_ig_counterparts(con, suggestions, flow_prefixes)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Attach `ig_recipe_id` to each suggestion: the intergovernmental-expenditure
|
||||||
|
#' recipe (an M-to-local or L-to-state recipe) whose component codes cover
|
||||||
|
#' exactly the same set of function suffixes as the firing recipe's own
|
||||||
|
#' components, e.g. `corrections_combined`'s {E04, E05} -> suffixes {"04",
|
||||||
|
#' "05"} matches `corrections_ig_local_combined`'s {M04, M05} -> the same
|
||||||
|
#' {"04", "05"}. `NULL` when no such recipe exists, which also covers the
|
||||||
|
#' case where the firing recipe already IS the IG recipe (self-matches are
|
||||||
|
#' excluded, so an IG recipe never names itself as its own counterpart).
|
||||||
|
#'
|
||||||
|
#' Matching is deliberately an exact set match, not "any suffix in common":
|
||||||
|
#' the two-digit suffix only means the same "function" across recipes that
|
||||||
|
#' share the underlying Census functional-classification scheme (E/F/G/L/M
|
||||||
|
#' all use "04"/"05" for corrections). M/L "combined other" codes (47/89/
|
||||||
|
#' 91-94) reuse digits for an unrelated catch-all construct, so e.g.
|
||||||
|
#' `general_gov_e89_wide`'s {E85, E89} -> {"85", "89"} must NOT match
|
||||||
|
#' `ige_local_m89_wide`'s {"89", "91", "92", "93"} on the shared "89" alone.
|
||||||
|
#' Checked by hand against the full harmonization_recipes catalog: only the
|
||||||
|
#' corrections family (E/F/G/M, suffixes 04/05) has an exact-set match in
|
||||||
|
#' this corpus.
|
||||||
|
#'
|
||||||
|
#' Exact-set suffix matching is NOT enough on its own, though: the same
|
||||||
|
#' reused-digit problem exists ACROSS the revenue-side IG families too.
|
||||||
|
#' `ig_local_d47_wide` (D47/D94, suffixes {"47","94"}) is an exact-set match
|
||||||
|
#' for `ige_local_m47_wide` (M47/M94, same suffixes) even though one is
|
||||||
|
#' intergovernmental REVENUE received from local governments and the other is
|
||||||
|
#' intergovernmental EXPENDITURE paid to local governments -- unrelated flows
|
||||||
|
#' that happen to reuse "47"/"94" for their own "transit/utilities" and
|
||||||
|
#' "other/combined" catch-alls. `ig_federal_b47_wide`, `ig_state_c47_wide`,
|
||||||
|
#' and their `*_89` siblings all collide the same way. None of this is
|
||||||
|
#' reachable via `cog_revenue()` in the bundled fixture today (its B/C/D
|
||||||
|
#' recipes never happen to have a covered gap year for any fixture govid),
|
||||||
|
#' but it IS reachable via a mis-scoped `cog_spending()` call on a
|
||||||
|
#' revenue-only category, e.g. `cog_spending(gov, category = "IG Federal")`
|
||||||
|
#' fires `ig_federal_b47_wide`/`ig_federal_b89_wide` for real in the fixture
|
||||||
|
#' -- so this is a live, not merely theoretical, gap.
|
||||||
|
#'
|
||||||
|
#' Two flow-family checks close this, both required (see
|
||||||
|
#' `tests/testthat/test-expenditure-concept.R`, "revenue-flavored ... never
|
||||||
|
#' receives an M/L counterpart" tests, for the pairwise verification):
|
||||||
|
#' 1. `own_prefix %in% flow_prefixes`: the firing recipe's own component
|
||||||
|
#' codes must belong to the calling verb's own flow family (the same
|
||||||
|
#' `flow_prefixes` `.build_harmonization_block()` uses, see
|
||||||
|
#' `R/basis.R`). This blocks a recipe surfaced through a mis-scoped
|
||||||
|
#' category from ever reaching the M/L search, e.g. `cog_spending()`'s
|
||||||
|
#' flow_prefixes are `c("E","F","G")`, which `ig_federal_b47_wide`'s own
|
||||||
|
#' `"B"` is not part of.
|
||||||
|
#' 2. `own_prefix %in% c("E","F","G")`: M/L only ever pairs with the
|
||||||
|
#' DIRECT-expenditure family, never with revenue (`cog_revenue()`'s
|
||||||
|
#' flow_prefixes already fold B/C/D in as ordinary revenue -- there is
|
||||||
|
#' no separate "Total" bolt-on for revenue the way `expenditure_concept`
|
||||||
|
#' adds one for spending) and never with ANOTHER M/L recipe (without
|
||||||
|
#' this check, `ige_local_m47_wide` would wrongly match sibling
|
||||||
|
#' `ige_state_l47_wide` on their shared {"47","94"} suffix set).
|
||||||
|
#' Condition 1 alone does not catch this: under `cog_revenue()`,
|
||||||
|
#' `ig_federal_b47_wide`'s own `"B"` IS inside revenue's own
|
||||||
|
#' `flow_prefixes`, so only this second, family-specific check blocks
|
||||||
|
#' the search.
|
||||||
|
#' @noRd
|
||||||
|
.attach_ig_counterparts <- function(con, suggestions, flow_prefixes) {
|
||||||
|
if (length(suggestions) == 0L) return(suggestions)
|
||||||
|
|
||||||
|
comp <- DBI::dbGetQuery(con,
|
||||||
|
"SELECT recipe_id, component_code FROM harmonization_recipes")
|
||||||
|
comp$prefix <- substr(comp$component_code, 1L, 1L)
|
||||||
|
comp$suffix <- substr(comp$component_code, 2L, nchar(comp$component_code))
|
||||||
|
suffix_sets <- lapply(split(comp$suffix, comp$recipe_id), function(x) sort(unique(x)))
|
||||||
|
prefix_sets <- lapply(split(comp$prefix, comp$recipe_id), function(x) sort(unique(x)))
|
||||||
|
|
||||||
|
ig_recipe_ids <- unique(comp$recipe_id[comp$prefix %in% c("M", "L")])
|
||||||
|
|
||||||
|
find_counterpart <- function(rid) {
|
||||||
|
own_prefix <- prefix_sets[[rid]]
|
||||||
|
own_suffix <- suffix_sets[[rid]]
|
||||||
|
if (is.null(own_prefix) || is.null(own_suffix)) return(NULL)
|
||||||
|
if (!all(own_prefix %in% flow_prefixes)) return(NULL)
|
||||||
|
if (!all(own_prefix %in% c("E", "F", "G"))) return(NULL)
|
||||||
|
for (cand in ig_recipe_ids) {
|
||||||
|
if (identical(cand, rid)) next
|
||||||
|
if (setequal(suffix_sets[[cand]], own_suffix)) return(cand)
|
||||||
|
}
|
||||||
|
NULL
|
||||||
|
}
|
||||||
|
|
||||||
|
lapply(suggestions, function(s) {
|
||||||
|
# `s$ig_recipe_id <- NULL` would DELETE the element rather than set it
|
||||||
|
# (standard R list-assignment gotcha), leaving no-match entries missing
|
||||||
|
# the key entirely instead of carrying it as NULL. Single-bracket
|
||||||
|
# assignment with a wrapped list preserves a NULL-valued element so the
|
||||||
|
# field is always present, per the brief's "NULL when there is none".
|
||||||
|
s["ig_recipe_id"] <- list(find_counterpart(s$recipe_id))
|
||||||
|
s
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Emit the single cli::cli_inform() message summarizing all suggestions
|
||||||
|
#' for a verb call (the brief's "one message", not one per suggestion).
|
||||||
|
#' Bullet text is pre-formatted plain text (no cli/glue `{}` markup) since
|
||||||
|
#' recipe ids/labels are untrusted-ish data values, not literal call-site
|
||||||
|
#' expressions. When a suggestion has an `ig_recipe_id`, one indented
|
||||||
|
#' continuation line is appended naming the intergovernmental counterpart
|
||||||
|
#' recipe (embedded `\n` renders as a hanging-indent continuation of the
|
||||||
|
#' same bullet under cli, not a new bullet). Same treatment for
|
||||||
|
#' `suppressed_amount` (uscogdata#9): only present when dollars were
|
||||||
|
#' actually measured as excluded (an `empty_year` fire can carry them too --
|
||||||
|
#' see `.build_suggestions()` -- so this keys off the amount, not `trigger`).
|
||||||
|
#' @noRd
|
||||||
|
.inform_suggestions <- function(suggestions) {
|
||||||
|
bullets <- vapply(suggestions, function(s) {
|
||||||
|
bullet <- sprintf("%s (%d-%d): %s", s$recipe_id,
|
||||||
|
s$available_years[1], s$available_years[2], s$hint)
|
||||||
|
# Only present when dollars were actually measured as excluded. An
|
||||||
|
# empty_year fire can carry them too -- the year had no rows AND the
|
||||||
|
# component was suppressed -- which is strictly more informative.
|
||||||
|
if (isTRUE(s$suppressed_amount > 0)) {
|
||||||
|
bullet <- paste0(bullet, sprintf(
|
||||||
|
"\n $%s excluded from %s (%s), published as an aggregate or outside the crosswalk",
|
||||||
|
formatC(s$suppressed_amount, format = "f", digits = 0, big.mark = ","),
|
||||||
|
paste0("FY", s$suppressed_years, collapse = ", "),
|
||||||
|
paste(s$suppressed_codes, collapse = ", ")))
|
||||||
|
}
|
||||||
|
if (!is.null(s$ig_recipe_id)) {
|
||||||
|
bullet <- paste0(bullet, sprintf(
|
||||||
|
"\n intergovernmental counterpart: recipe = '%s'", s$ig_recipe_id))
|
||||||
|
}
|
||||||
|
bullet
|
||||||
|
}, character(1))
|
||||||
|
cli::cli_inform(c(
|
||||||
|
i = "Incomplete coverage for the requested years; a harmonization recipe may fill it:",
|
||||||
|
stats::setNames(bullets, rep("*", length(bullets)))
|
||||||
|
))
|
||||||
|
}
|
||||||
+117
@@ -0,0 +1,117 @@
|
|||||||
|
# R/suppression.R
|
||||||
|
# Split out of R/suggestions.R (2026-08-05) to keep files under the project's
|
||||||
|
# 400-line limit. Owns the second qualifying path for coverage signposting
|
||||||
|
# (uscogdata#9): measuring, per government, the component dollars the
|
||||||
|
# calling verb's own long view structurally excludes (aggregate-published,
|
||||||
|
# or absent from summary_categories). See R/suggestions.R for the
|
||||||
|
# orchestrator (`.build_suggestions()`) that calls this and the full
|
||||||
|
# uscogdata#9 background.
|
||||||
|
|
||||||
|
#' Measure, per (recipe, year), the component dollars this government holds
|
||||||
|
#' that the calling verb's own long view structurally excludes.
|
||||||
|
#'
|
||||||
|
#' This is the second qualifying path for a suggestion (uscogdata#9). The
|
||||||
|
#' first -- row absence -- only fires when a category returns NOTHING in a
|
||||||
|
#' requested year, which is how Corrections behaves in the wide era. Public
|
||||||
|
#' Welfare is the failure mode it misses: E74/E75/E77/E79 still return rows,
|
||||||
|
#' so there is no absence to detect, while E67/E68 (aggregate-flagged 1967-
|
||||||
|
#' 2011, and absent from `summary_categories` entirely) are dropped. The
|
||||||
|
#' caller gets a plausible number a third too low, silently.
|
||||||
|
#'
|
||||||
|
#' "Structurally excluded" is decided by anti-joining the verb's REAL long
|
||||||
|
#' view rather than restating its WHERE clause, so this stays correct if
|
||||||
|
#' `spending_long_harmonized` / `revenue_long_harmonized` ever change. That
|
||||||
|
#' anti-join is keyed on `item_code`, which is sound only because
|
||||||
|
#' harmonization never renames a recipe component -- asserted by the "no
|
||||||
|
#' recipe component is ever renamed by harmonization" test in
|
||||||
|
#' tests/testthat/test-recipes.R.
|
||||||
|
#'
|
||||||
|
#' Note what this deliberately does NOT count as suppressed: a component
|
||||||
|
#' excluded from the RESULT for scoping reasons -- because it belongs to a
|
||||||
|
#' different `category`, or because `expenditure_concept` narrowed the
|
||||||
|
#' subtypes -- is still present in the view, so it never fires. Suggesting a
|
||||||
|
#' recipe is a coverage fix, not a category redefinition.
|
||||||
|
#'
|
||||||
|
#' `flow_prefixes` (uscogdata#9 review, finding I1) restricts the measured
|
||||||
|
#' components to the CALLING VERB's own flow family (`c("E","F","G")` for
|
||||||
|
#' spending, `c("T","A","U","B","C","D")` for revenue). Without this, a
|
||||||
|
#' candidate recipe belonging to the OTHER flow family is always absent from
|
||||||
|
#' this verb's view (by construction -- `cog_revenue()`'s view never carries
|
||||||
|
#' an E-coded row) and so was always reported as "suppressed", fabricating a
|
||||||
|
#' dollar claim across flow families (`cog_revenue(category = "Corrections")`
|
||||||
|
#' claimed $3.63B excluded that `cog_spending()` reports and fully accounts
|
||||||
|
#' for). Filtering on `LEFT(r.component_code, 1)` also drops M/L-prefixed
|
||||||
|
#' components from measurement under `cog_spending()` (`flow_prefixes` never
|
||||||
|
#' includes "M"/"L") -- harmless today, because a recipe's own M/L components
|
||||||
|
#' (e.g. `corrections_ig_local_combined`'s M04/M05) are present in the view
|
||||||
|
#' in every year they exist and so never fired as suppressed anyway, but
|
||||||
|
#' worth recording since this filter is now the thing relied on to prevent
|
||||||
|
#' it.
|
||||||
|
#'
|
||||||
|
#' @param con Active DuckDB connection.
|
||||||
|
#' @param candidates Character vector of recipe ids to measure.
|
||||||
|
#' @param cohort The verb's cohort object (see `.make_cohort()`), rendered
|
||||||
|
#' into the govid predicate on both the outer scan and the restated
|
||||||
|
#' NOT EXISTS filter.
|
||||||
|
#' @param years Integer vector of requested years.
|
||||||
|
#' @param long_view Name of the verb's long view, from `.select_long_view()`.
|
||||||
|
#' @param flow_prefixes The calling verb's own flow-type prefixes (see
|
||||||
|
#' `.build_suggestions()`). Only recipe components whose first character is
|
||||||
|
#' in this set are measured.
|
||||||
|
#' @return Tibble of `recipe_id`, `year`, `suppressed_amount` (full US
|
||||||
|
#' dollars), `suppressed_codes` (comma-joined, sorted). Zero rows when
|
||||||
|
#' nothing is suppressed.
|
||||||
|
#' @noRd
|
||||||
|
.suppressed_components <- function(con, candidates, cohort, years, long_view,
|
||||||
|
flow_prefixes) {
|
||||||
|
empty <- tibble::tibble(
|
||||||
|
recipe_id = character(0), year = numeric(0),
|
||||||
|
suppressed_amount = numeric(0), suppressed_codes = character(0)
|
||||||
|
)
|
||||||
|
if (length(candidates) == 0L) return(empty)
|
||||||
|
|
||||||
|
# long_view is interpolated as a SQL IDENTIFIER, not a literal, so it can
|
||||||
|
# never be quoted safely. It is always internally derived from a fixed
|
||||||
|
# view_base, so an off-allowlist value is a programming error, not input.
|
||||||
|
if (!long_view %in% c("spending_long", "spending_long_harmonized",
|
||||||
|
"revenue_long", "revenue_long_harmonized")) {
|
||||||
|
cli::cli_abort(
|
||||||
|
"Internal error: unexpected `long_view` {.val {long_view}}.",
|
||||||
|
class = "uscogdata_internal_error"
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
sql <- sprintf(
|
||||||
|
"SELECT r.recipe_id,
|
||||||
|
l.year,
|
||||||
|
SUM(l.amt) * 1000.0 AS suppressed_amount,
|
||||||
|
string_agg(DISTINCT l.item_code, ',' ORDER BY l.item_code)
|
||||||
|
AS suppressed_codes
|
||||||
|
FROM long l
|
||||||
|
JOIN harmonization_recipes r
|
||||||
|
ON l.item_code = r.component_code
|
||||||
|
AND l.year BETWEEN r.year_min AND r.year_max
|
||||||
|
AND (r.gov_type_scope = 'all'
|
||||||
|
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||||
|
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||||
|
WHERE r.recipe_id IN (%1$s)
|
||||||
|
AND %2$s
|
||||||
|
AND l.year IN (%3$s)
|
||||||
|
AND l.amt <> 0
|
||||||
|
AND LEFT(r.component_code, 1) IN (%5$s)
|
||||||
|
AND NOT EXISTS (
|
||||||
|
SELECT 1 FROM %4$s v
|
||||||
|
WHERE v.canonical_govid = l.canonical_govid
|
||||||
|
AND v.year = l.year
|
||||||
|
AND v.item_code = l.item_code
|
||||||
|
AND v.year IN (%3$s) -- restated: enables partition pruning (I3a)
|
||||||
|
AND %6$s -- restated: pushes the cohort filter (I3a)
|
||||||
|
)
|
||||||
|
GROUP BY 1, 2
|
||||||
|
ORDER BY 1, 2",
|
||||||
|
.sql_lit_chr(candidates), .cohort_sql(cohort, "l.canonical_govid"),
|
||||||
|
paste(as.integer(years), collapse = ","), long_view,
|
||||||
|
.sql_lit_chr(flow_prefixes), .cohort_sql(cohort, "v.canonical_govid")
|
||||||
|
)
|
||||||
|
tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||||
|
}
|
||||||
@@ -1,13 +1,162 @@
|
|||||||
# R/views.R
|
# R/views.R
|
||||||
|
|
||||||
|
# SQL files that cannot be registered unconditionally against a v4 corpus,
|
||||||
|
# for one of two distinct reasons -- both fail at CREATE VIEW time (DuckDB
|
||||||
|
# resolves a view's source schema eagerly, even though it defers execution),
|
||||||
|
# so a v4 corpus can't tolerate either unconditionally:
|
||||||
|
#
|
||||||
|
# (a) Missing FILE. 33-/34-/35- read_parquet() a v5-only parquet table
|
||||||
|
# (harmonization_map.parquet, harmonization_recipes.parquet,
|
||||||
|
# series_breaks.parquet) that doesn't exist at all on a v4 corpus --
|
||||||
|
# "IO Error: No files found".
|
||||||
|
#
|
||||||
|
# (b) Missing COLUMN. 22-/23-/25- reference `long.harmonized_code`, a
|
||||||
|
# column that does not exist on a v4 corpus's `long` table (harmonized
|
||||||
|
# space was introduced in schema v5) -- "Binder Error: Referenced
|
||||||
|
# column harmonized_code not found". 42-/43-/45- are on this list only
|
||||||
|
# because they SELECT s.* FROM the (a)/(b) views above, so they'd fail
|
||||||
|
# to resolve their own source view if it weren't already skipped.
|
||||||
|
#
|
||||||
|
# Registration is therefore gated on manifest$schema_version >= 5 for all of
|
||||||
|
# them; verb-level *usage* of the resulting views is separately gated by
|
||||||
|
# .resolve_basis() / .require_schema_v5().
|
||||||
|
.harmonization_view_files <- c(
|
||||||
|
"22-spending_long_harmonized.sql",
|
||||||
|
"23-revenue_long_harmonized.sql",
|
||||||
|
"25-ig_long_harmonized.sql",
|
||||||
|
"33-harmonization_map.sql",
|
||||||
|
"34-harmonization_recipes.sql",
|
||||||
|
"35-series_breaks_pq.sql",
|
||||||
|
"42-spending_annotated_harmonized.sql",
|
||||||
|
"43-revenue_annotated_harmonized.sql",
|
||||||
|
"45-ig_annotated_harmonized.sql"
|
||||||
|
)
|
||||||
|
|
||||||
|
# The representation contract (cog_pipeline#64): two parquet tables that say
|
||||||
|
# what an ABSENT cell means in a given year. Gated on manifest PRESENCE, not
|
||||||
|
# on schema_version, because the sparsification that introduced them did not
|
||||||
|
# bump the version -- the pre-sparsification corpus this package shipped
|
||||||
|
# against until 2026-07-30 was already schema v6 and carried neither table.
|
||||||
|
# Keying off the version number would therefore register a view over a file
|
||||||
|
# that does not exist and fail at CREATE VIEW time on exactly the corpora this
|
||||||
|
# check exists to tolerate.
|
||||||
|
.representation_view_files <- c(
|
||||||
|
"36-representation.sql" = "representation.parquet",
|
||||||
|
"37-code_set.sql" = "code_set.parquet"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Cash and security holdings (uscogdata#25). 46- selects
|
||||||
|
# `c.balance_subtype`, a column that arrived with cog_pipeline #76/#77 and
|
||||||
|
# WITHOUT a schema_version bump -- so neither existing gate applies:
|
||||||
|
# .harmonization_view_files keys on schema_version, .representation_view_files
|
||||||
|
# on the presence of a FILE. Here the discriminator is a COLUMN on a table
|
||||||
|
# that exists either way. CREATE VIEW resolves its source schema eagerly, so
|
||||||
|
# on an older corpus 46- would fail at registration with "Binder Error:
|
||||||
|
# Referenced column balance_subtype not found" rather than at query time.
|
||||||
|
.balance_view_files <- c("26-balance_long.sql", "46-balance_annotated.sql")
|
||||||
|
|
||||||
|
#' Does the mounted corpus's `summary_categories` carry `balance_subtype`?
|
||||||
|
#' Probed against the live connection rather than the manifest, because the
|
||||||
|
#' manifest describes files, not columns.
|
||||||
|
#' @noRd
|
||||||
|
.corpus_has_balance_subtype <- function(con) {
|
||||||
|
n <- DBI::dbGetQuery(con,
|
||||||
|
"SELECT COUNT(*) AS n FROM information_schema.columns
|
||||||
|
WHERE table_name = 'summary_categories'
|
||||||
|
AND column_name = 'balance_subtype'"
|
||||||
|
)$n
|
||||||
|
isTRUE(as.integer(n) > 0L)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Does the mounted corpus publish `file` (e.g. "code_set.parquet")?
|
||||||
|
#' Reads the manifest's metadata list rather than stat-ing the URL, so it
|
||||||
|
#' works identically for a local fixture and a remote share.
|
||||||
|
#' @noRd
|
||||||
|
.corpus_has_table <- function(manifest, file) {
|
||||||
|
paths <- vapply(manifest$files$metadata %||% list(),
|
||||||
|
function(f) as.character(f$path %||% ""), character(1))
|
||||||
|
file %in% basename(paths)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Build the SQL path expression for the partitioned `long` table.
|
||||||
|
#'
|
||||||
|
#' DuckDB cannot expand a glob over generic HTTP: there is no directory
|
||||||
|
#' listing to expand against, and `allow_asterisks_in_http_paths` only
|
||||||
|
#' forwards the literal `**/*` as a filename, which 404s. Measured against
|
||||||
|
#' the published corpus on 2026-08-08, an explicit file list returns the
|
||||||
|
#' same 46,148,034 rows the (working) `hf://` glob does, and
|
||||||
|
#' `hive_partitioning = true` still recovers `year` from the paths.
|
||||||
|
#'
|
||||||
|
#' The manifest already enumerates every partition, so we build the list
|
||||||
|
#' from it. This is host-agnostic -- Nextcloud, HuggingFace and a local
|
||||||
|
#' fixture take the same path -- where an `hf://` URL would tie the reader
|
||||||
|
#' to one vendor's protocol and still need special-casing, since manifest
|
||||||
|
#' fetching goes through httr2, which cannot speak `hf://`.
|
||||||
|
#'
|
||||||
|
#' Falls back to the glob when the manifest carries no partition list: a
|
||||||
|
#' hand-built manifest in a test (see test-views.R) or a corpus predating
|
||||||
|
#' the field. Both are local, where globbing works.
|
||||||
|
#' @noRd
|
||||||
|
.long_files_sql <- function(url, manifest) {
|
||||||
|
parts <- manifest$files$long_partitions %||% list()
|
||||||
|
if (length(parts) == 0L) {
|
||||||
|
return(.sql_lit_chr(paste0(url, "data/long/**/*.parquet")))
|
||||||
|
}
|
||||||
|
paths <- vapply(parts, function(p) as.character(p$path), character(1))
|
||||||
|
paste0("[", .sql_lit_chr(paste0(url, paths)), "]")
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Substitute the corpus-location tokens in a view's SQL text.
|
||||||
|
#'
|
||||||
|
#' One place knows the token vocabulary. `.register_views()` and the tests
|
||||||
|
#' that execute a view file directly both route through here. This exists
|
||||||
|
#' because four test sites had hand-rolled the `{url}` substitution -- one
|
||||||
|
#' of them commented as doing it "exactly as .register_views() does" -- and
|
||||||
|
#' every one of them broke the moment a second token was introduced.
|
||||||
|
#'
|
||||||
|
#' `{long_files}` must be substituted BEFORE `{url}`: it expands to a string
|
||||||
|
#' that itself contains the url, so the reverse order leaves the token in
|
||||||
|
#' place and DuckDB's parser fails on the brace.
|
||||||
|
#'
|
||||||
|
#' `manifest` defaults to empty, which routes `.long_files_sql()` to its glob
|
||||||
|
#' fallback -- correct for the local temp corpora the direct-execution tests
|
||||||
|
#' build.
|
||||||
|
#' @noRd
|
||||||
|
#' `fixed = TRUE` is load-bearing, not a style choice.
|
||||||
|
#'
|
||||||
|
#' In regex mode, `gsub()` interprets backslashes in the REPLACEMENT string as
|
||||||
|
#' escape sequences and silently drops them. A Windows corpus path is full of
|
||||||
|
#' them, so `C:\Users\RUNNER\AppData\...` was substituted in as
|
||||||
|
#' `C:UsersRUNNERAppData...` and every DuckDB read failed with "No files found
|
||||||
|
#' that match the pattern". `fixed = TRUE` treats pattern and replacement as
|
||||||
|
#' literal text, which is what a filesystem path needs.
|
||||||
|
#'
|
||||||
|
#' This is why the package could not read a LOCAL corpus on Windows at all --
|
||||||
|
#' including the test fixture, hence the entire suite, and any `cog_mirror()`
|
||||||
|
#' copy. Remote https URLs were unaffected, having no backslashes, which is
|
||||||
|
#' part of why it stayed hidden: the bug predates the `{long_files}` token and
|
||||||
|
#' lived in the original `{url}` substitution, unnoticed because nothing ever
|
||||||
|
#' ran on Windows until the mirror's check matrix existed.
|
||||||
|
#' @noRd
|
||||||
|
.render_view_sql <- function(sql, url, manifest = list()) {
|
||||||
|
sql <- gsub("{long_files}", .long_files_sql(url, manifest), sql, fixed = TRUE)
|
||||||
|
gsub("{url}", url, sql, fixed = TRUE)
|
||||||
|
}
|
||||||
|
|
||||||
#' Register DuckDB views from inst/sql/ SQL files
|
#' Register DuckDB views from inst/sql/ SQL files
|
||||||
#' @noRd
|
#' @noRd
|
||||||
.register_views <- function(con, url, manifest) {
|
.register_views <- function(con, url, manifest) {
|
||||||
sql_dir <- system.file("sql", package = "uscogdata")
|
sql_dir <- system.file("sql", package = "uscogdata")
|
||||||
files <- list.files(sql_dir, pattern = "\\.sql$", full.names = TRUE)
|
files <- sort(list.files(sql_dir, pattern = "\\.sql$", full.names = TRUE))
|
||||||
|
schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L))
|
||||||
for (f in files) {
|
for (f in files) {
|
||||||
|
base <- basename(f)
|
||||||
|
if (base %in% .harmonization_view_files && schema_version < 5L) next
|
||||||
|
if (base %in% names(.representation_view_files) &&
|
||||||
|
!.corpus_has_table(manifest, .representation_view_files[[base]])) next
|
||||||
|
if (base %in% .balance_view_files && !.corpus_has_balance_subtype(con)) next
|
||||||
sql <- paste(readLines(f, warn = FALSE), collapse = "\n")
|
sql <- paste(readLines(f, warn = FALSE), collapse = "\n")
|
||||||
sql <- gsub("\\{url\\}", url, sql, fixed = FALSE)
|
sql <- .render_view_sql(sql, url, manifest)
|
||||||
DBI::dbExecute(con, sql)
|
DBI::dbExecute(con, sql)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,60 +1,258 @@
|
|||||||
# uscogdata
|
# uscogdata
|
||||||
|
|
||||||
Curated R reader for the Civilytics US Census of Governments finance corpus.
|
<!-- badges: start -->
|
||||||
|
[](https://civilytics.r-universe.dev/uscogdata)
|
||||||
|
[](LICENSE.md)
|
||||||
|
<!-- badges: end -->
|
||||||
|
|
||||||
Provides unit-level financial profiles, geographic rollups, and peer comparisons
|
A curated R reader for the Civilytics US Census of Governments finance corpus —
|
||||||
with auditable provenance and built-in cross-vintage correctness. Reads the
|
every dollar that US state, county, municipal and township governments reported
|
||||||
published corpus (Hive-partitioned parquet + manifest.json) directly from
|
raising and spending, from **FY1967 to FY2024**, in one queryable place.
|
||||||
Nextcloud via DuckDB httpfs — no local bulk downloads required.
|
|
||||||
|
|
||||||
## Status
|
The Census of Governments is the only nationwide source for local government
|
||||||
|
finance, and it is hard to use: item codes change meaning across vintages,
|
||||||
|
government identifiers were renumbered in 2017, and an absent value means
|
||||||
|
"published zero" in one era and "not reported" in the next. This package
|
||||||
|
handles each of those problems, and it tells you when it has — every result
|
||||||
|
carries provenance describing what was converted, what was aggregated, and
|
||||||
|
which known series breaks intersect your query.
|
||||||
|
|
||||||
Under active development (Phase 2 of the cog_pipeline project). See
|
**Scope:** government types 0–3 (state, county, municipality, township).
|
||||||
`../cog_pipeline/docs/reader-specification.md` for the reader contract this
|
56 fiscal years, 46,148,034 rows, 190.6 MB. There is no source data for FY1968
|
||||||
package implements.
|
or FY1969. Special districts (type 4) and school districts (type 5) are
|
||||||
|
excluded pending validation.
|
||||||
|
|
||||||
## Installation
|
## Where the data comes from
|
||||||
|
|
||||||
|
The corpus is published and documented at the **[US Census of Governments
|
||||||
|
Finance API](https://pages.civilytics.org/cog-api/)**. Start there for how the
|
||||||
|
data was built, how the identifier and item-code reconciliation works, and what
|
||||||
|
the corpus does and does not cover.
|
||||||
|
|
||||||
|
- **[API documentation and walkthroughs](https://pages.civilytics.org/cog-api/)**
|
||||||
|
— reference, data dictionary, and worked examples such as the
|
||||||
|
[Southern states guide](https://pages.civilytics.org/cog-api/cog-api-south-guide.html)
|
||||||
|
- **[Live API](https://cog-api.civilytics.org/api/v1/)** — the same corpus over
|
||||||
|
HTTP, for Tableau, Python, or anything that isn't R
|
||||||
|
- **[Bulk corpus on Hugging Face](https://huggingface.co/datasets/civilytics/us-cog-finance)**
|
||||||
|
— CC-BY-4.0; the same parquet files this package reads
|
||||||
|
- **[Census Bureau source data](https://www.census.gov/programs-surveys/gov-finances.html)**
|
||||||
|
— the underlying public files
|
||||||
|
|
||||||
|
## Install
|
||||||
|
|
||||||
```r
|
```r
|
||||||
# pak::pkg_install("gitea.civilytics.org/Civilytics/uscogdata")
|
install.packages("uscogdata",
|
||||||
|
repos = c("https://civilytics.r-universe.dev",
|
||||||
|
"https://cloud.r-project.org"))
|
||||||
```
|
```
|
||||||
|
|
||||||
## Configuration
|
Or from source:
|
||||||
|
|
||||||
- `USCOGDATA_URL` — corpus root URL (public Nextcloud share, trailing slash)
|
|
||||||
- `USCOGDATA_CACHE_DIR` — optional override for the manifest cache directory
|
|
||||||
- `USCOGDATA_MANIFEST_TTL_SECS` — optional manifest re-fetch TTL (default 3600)
|
|
||||||
|
|
||||||
## Developer notes
|
|
||||||
|
|
||||||
### Testing
|
|
||||||
|
|
||||||
The package ships a bundled fixture corpus at `inst/extdata/fixture_corpus/` —
|
|
||||||
a 3.6 MB two-year slice (2019 + 2020) of the full corpus covering all 50
|
|
||||||
states. `tests/testthat/setup.R` automatically points `USCOGDATA_URL` at this
|
|
||||||
fixture, so the full test suite runs offline with no network dependency:
|
|
||||||
|
|
||||||
```r
|
```r
|
||||||
devtools::test() # uses bundled fixture, no credentials required
|
pak::pkg_install("git::https://gitea.civilytics.org/Civilytics/uscogdata.git")
|
||||||
```
|
```
|
||||||
|
|
||||||
### Releasing against the live corpus
|
## Quickstart
|
||||||
|
|
||||||
Before cutting a release, run the test suite against the published corpus to
|
No configuration, no credentials, no download. The package reads the published
|
||||||
catch any drift between the fixture and the real data:
|
corpus over HTTPS by default.
|
||||||
|
|
||||||
```r
|
```r
|
||||||
Sys.setenv(USCOGDATA_URL = "<published-corpus-url-with-trailing-slash>")
|
library(uscogdata)
|
||||||
devtools::test()
|
|
||||||
|
# Resolve a place name to a canonical government id
|
||||||
|
madison <- cog_gov_search(name = "Madison", state = "WI", type = 2)
|
||||||
|
madison$canonical_govid
|
||||||
|
#> [1] "552025209777"
|
||||||
|
|
||||||
|
# Police spending, inflation-adjusted and per capita
|
||||||
|
spend <- cog_spending(
|
||||||
|
madison$canonical_govid,
|
||||||
|
years = 2012:2022,
|
||||||
|
category = "Police",
|
||||||
|
per_capita = TRUE,
|
||||||
|
adjust_to_year = 2023
|
||||||
|
)
|
||||||
|
|
||||||
|
# What did that result do to the numbers, and what should you know about them?
|
||||||
|
cog_explain(spend)
|
||||||
```
|
```
|
||||||
|
|
||||||
When the live-corpus run is clean, strip the fixture from the built package by
|
`years` is required — there is no implicit full-history default.
|
||||||
adding this line to `.Rbuildignore`:
|
|
||||||
|
|
||||||
```
|
## Two ways to read the corpus
|
||||||
^inst/extdata/fixture_corpus$
|
|
||||||
|
| | Remote (default) | Mirrored |
|
||||||
|
|---|---|---|
|
||||||
|
| Setup | none | `cog_mirror(dest)`, 190.6 MB once |
|
||||||
|
| Disk used | **0 MB** — HTTP range requests only | 190.6 MB |
|
||||||
|
| Per query | ~4 s (one government, one year)<br>~6 s (one government, 23 years) | local speed |
|
||||||
|
| Good for | trying it out, teaching, one-off questions | repeated analysis, offline work, reproducibility |
|
||||||
|
|
||||||
|
Nothing is written to disk in remote mode: DuckDB fetches the parquet footer,
|
||||||
|
works out which row groups it needs, and reads only those. Nothing is cached
|
||||||
|
between sessions either, so every query goes back to the network.
|
||||||
|
|
||||||
|
The default points at a public HuggingFace mirror of the corpus. If you would
|
||||||
|
rather not depend on a third party — for reproducibility, for an air-gapped
|
||||||
|
environment, or on principle — **the escape hatch is one function call**:
|
||||||
|
|
||||||
|
```r
|
||||||
|
cog_mirror("~/cog-corpus")
|
||||||
|
Sys.setenv(USCOGDATA_URL = "~/cog-corpus/")
|
||||||
```
|
```
|
||||||
|
|
||||||
The test suite is URL-agnostic — `setup.R` falls back to `USCOGDATA_URL` when
|
After that, nothing in your analysis touches an external service.
|
||||||
the bundled fixture is absent, so no test code changes are needed for the
|
|
||||||
release run or after stripping the fixture.
|
### Configuration
|
||||||
|
|
||||||
|
- `USCOGDATA_URL` — corpus root: an HTTPS URL or a local path, **trailing slash required**
|
||||||
|
- `USCOGDATA_CACHE_DIR` — where the manifest is cached (default: user cache dir)
|
||||||
|
- `USCOGDATA_MANIFEST_TTL_SECS` — manifest re-fetch interval (default 3600)
|
||||||
|
- `USCOGDATA_DUCKDB_THREADS` — cap DuckDB's thread count (default: every visible core)
|
||||||
|
- `USCOGDATA_DUCKDB_MEMORY_LIMIT` — cap DuckDB's memory, e.g. `"4GB"` (default: DuckDB's own)
|
||||||
|
|
||||||
|
Each also has an `options()` spelling — `uscogdata.url`, `uscogdata.duckdb_threads`,
|
||||||
|
and so on — and the environment variable wins where both are set.
|
||||||
|
|
||||||
|
The two DuckDB caps exist for **servers**, not laptops. Unset, DuckDB claims every
|
||||||
|
core it can see, which is right for one interactive session on your own machine and
|
||||||
|
wrong when several readers share a box: each claims the whole machine and they fight.
|
||||||
|
Capping costs roughly 5% on a single query and is worth it anywhere the process is
|
||||||
|
sharing hardware.
|
||||||
|
|
||||||
|
## Amounts are in full US dollars
|
||||||
|
|
||||||
|
Every amount column this package returns — `amt_nominal`, `amt_real`,
|
||||||
|
`amt_per_capita_nominal`, `amt_per_capita_real` — is in **full US dollars**.
|
||||||
|
|
||||||
|
The raw Census source files report **thousands of dollars**, and the corpus's
|
||||||
|
own `amt` column preserves that. The verbs multiply by 1000 on the way out, so
|
||||||
|
you never have to. The conversion is recorded in every result:
|
||||||
|
|
||||||
|
```r
|
||||||
|
attr(spend, "provenance")$transformations$units_conversion
|
||||||
|
#> $applied TRUE
|
||||||
|
#> $source_unit "$1,000s (raw Census)"
|
||||||
|
#> $target_unit "$USD"
|
||||||
|
#> $multiplier 1000
|
||||||
|
```
|
||||||
|
|
||||||
|
**Do not multiply again.** If you have read elsewhere that COG amounts are in
|
||||||
|
`$1,000s` — which is true of the raw Census files and of the corpus's own `amt`
|
||||||
|
column — that rule does not apply to anything a `cog_*()` verb hands you.
|
||||||
|
Applying it twice overstates every figure by 1000x, and the result looks
|
||||||
|
plausible rather than obviously wrong.
|
||||||
|
|
||||||
|
## Concepts worth understanding before you publish a number
|
||||||
|
|
||||||
|
### Primary vs Direct vs Total spending
|
||||||
|
|
||||||
|
`cog_spending(..., expenditure_concept = c("primary", "direct", "total"))`
|
||||||
|
controls *whose* spending a result counts. Concepts are defined as sets of the
|
||||||
|
crosswalk's `spend_subtype` values, never item-code first letters — the letter
|
||||||
|
`Y` alone spans revenue, expenditure and balance codes.
|
||||||
|
|
||||||
|
- **`"primary"`** (default) — the government's own service provision: current
|
||||||
|
operations, capital outlay, assistance payments.
|
||||||
|
- **`"direct"`** — Census's published Direct Expenditure: `primary` plus
|
||||||
|
interest on debt and insurance trust benefits (e.g. pensions).
|
||||||
|
- **`"total"`** — adds the intergovernmental leg, money handed to other
|
||||||
|
governments to spend. Meaningful for one government's own budget over time,
|
||||||
|
but it double-counts when summed across governments: a state's payment to a
|
||||||
|
county is the same dollar the county reports as its own direct spending.
|
||||||
|
|
||||||
|
**Rule of thumb: any figure spanning more than one government uses `primary`
|
||||||
|
or `direct`.** `cog_geographic_rollup()` and `cog_peer_compare()` enforce that
|
||||||
|
by refusing `"total"` outright. Worked examples in
|
||||||
|
`vignette("total-spending", package = "uscogdata")`.
|
||||||
|
|
||||||
|
### General vs Total revenue
|
||||||
|
|
||||||
|
`cog_revenue(..., revenue_concept = c("general", "total"))`:
|
||||||
|
|
||||||
|
- **`"general"`** (default) — Census General Revenue: own-source taxes,
|
||||||
|
charges and miscellaneous, plus federal, state and local aid.
|
||||||
|
- **`"total"`** — General plus utility revenue (`A91`–`A94`), liquor store
|
||||||
|
revenue (`A90`), and insurance trust revenue.
|
||||||
|
|
||||||
|
Census defines these by its own identity:
|
||||||
|
|
||||||
|
```
|
||||||
|
Total Revenue = General + Utility + Liquor Store + Insurance Trust
|
||||||
|
```
|
||||||
|
|
||||||
|
Two things to know before switching to `"total"`. **Utility revenue is large
|
||||||
|
for cities** — measured on the bundled fixture, utility plus liquor store is
|
||||||
|
15.9% of city revenue, against 1.2% for states and 1.7% for counties. And the
|
||||||
|
**employee-retirement (`X`) codes stop at FY2016**, when those systems moved to
|
||||||
|
the separate Annual Survey of Public Pensions, so a `"total"` series steps down
|
||||||
|
at the FY2016/FY2017 boundary for reasons of collection scope, not revenue
|
||||||
|
(series breaks `SB197`–`SB209`).
|
||||||
|
|
||||||
|
### Reporting coverage: the Census is only sometimes a census
|
||||||
|
|
||||||
|
**The Census of Governments is a complete enumeration only in years ending in
|
||||||
|
2 and 7.** Every other year is a sample, and the sample varies enormously —
|
||||||
|
measured on the bundled fixture, Wisconsin's 608-city universe rolls up 597
|
||||||
|
governments in FY2012 and 112 in FY2019.
|
||||||
|
|
||||||
|
A statewide total resting on a fifth of the universe looks exactly like one
|
||||||
|
resting on all of it, so every multi-government result now says which it is:
|
||||||
|
|
||||||
|
```r
|
||||||
|
attr(rollup, "provenance")$coverage # per-year n_units_reporting, is_census_year
|
||||||
|
```
|
||||||
|
|
||||||
|
`cog_geographic_rollup()`, `cog_peer_compare()` and `cog_find_peers()` take a
|
||||||
|
`coverage` argument — `"all"` (default), `"census"` (census years only), or
|
||||||
|
`"consistent"` (only units reporting in every requested year, a balanced
|
||||||
|
panel).
|
||||||
|
|
||||||
|
`n_units_reporting` is **category-conditional**, and it is not a response rate. A government that was surveyed and genuinely spends
|
||||||
|
nothing in the requested category is indistinguishable from one never surveyed.
|
||||||
|
|
||||||
|
### Absent cells mean two different things
|
||||||
|
|
||||||
|
Before FY2012, an absent cell means Census published `$0`. From FY2012 on, it
|
||||||
|
means not reported. `cog_spending(..., complete = TRUE)` fills the requested
|
||||||
|
grid and labels every row with which it is, via `value_source`:
|
||||||
|
|
||||||
|
| `value_source` | meaning | `amt_nominal` |
|
||||||
|
|---|---|---|
|
||||||
|
| `reported` | the corpus carries this cell | as published |
|
||||||
|
| `census_zero` | dense-source year (≤ FY2011), absent — Census published `$0` | `0` |
|
||||||
|
| `not_reported` | sparse-source year (≥ FY2012), absent — unknown | `NA` |
|
||||||
|
|
||||||
|
That `NA` is deliberate. Filling a modern absence with `0` would invent data.
|
||||||
|
|
||||||
|
### Series breaks surface on their own
|
||||||
|
|
||||||
|
Catalogued breaks that intersect your query appear in provenance whether or not
|
||||||
|
you went looking for them — `series_break_refs` for breaks in a specific item code, and
|
||||||
|
`corpus_break_refs` for caveats about the corpus as a whole (dollar precision
|
||||||
|
across the 1976/1977 boundary, the FY2017 identifier change, the FY2012
|
||||||
|
dense→sparse representation change). `cog_explain()` prints both.
|
||||||
|
|
||||||
|
## How to cite
|
||||||
|
|
||||||
|
```r
|
||||||
|
citation("uscogdata")
|
||||||
|
```
|
||||||
|
|
||||||
|
The corpus itself is published under CC-BY-4.0. Cite it as:
|
||||||
|
|
||||||
|
> Civilytics Consulting. US Census of Governments finance corpus.
|
||||||
|
> https://huggingface.co/datasets/civilytics/us-cog-finance
|
||||||
|
|
||||||
|
## Contributing
|
||||||
|
|
||||||
|
Development happens on [Gitea](https://gitea.civilytics.org/Civilytics/uscogdata);
|
||||||
|
[GitHub](https://github.com/civilytics/uscogdata) is a mirror that accepts
|
||||||
|
issues and pull requests. See [CONTRIBUTING.md](CONTRIBUTING.md) for how a
|
||||||
|
patch gets from there to here.
|
||||||
|
|
||||||
|
## License
|
||||||
|
|
||||||
|
MIT © Civilytics Consulting LLC. See [LICENSE.md](LICENSE.md).
|
||||||
|
|||||||
+27
-5
@@ -1,19 +1,41 @@
|
|||||||
url: ~
|
url: https://civilytics.r-universe.dev/uscogdata
|
||||||
|
|
||||||
template:
|
template:
|
||||||
bootstrap: 5
|
bootstrap: 5
|
||||||
|
|
||||||
reference:
|
reference:
|
||||||
|
- title: Financial data
|
||||||
|
desc: Spending, revenue and balance-sheet holdings for one or more governments.
|
||||||
|
contents:
|
||||||
|
- cog_spending
|
||||||
|
- cog_revenue
|
||||||
|
- cog_balances
|
||||||
- title: Search & basket
|
- title: Search & basket
|
||||||
desc: Resolve place names into canonical govids.
|
desc: Resolve place names into canonical govids.
|
||||||
contents:
|
contents:
|
||||||
- cog_gov_search
|
- cog_gov_search
|
||||||
- cog_basket_resolution
|
- cog_basket_resolution
|
||||||
- cog_basket_unresolved
|
- cog_basket_unresolved
|
||||||
- title: Session
|
- title: Comparison & aggregation
|
||||||
|
desc: Peer cohorts and geographic aggregates.
|
||||||
contents:
|
contents:
|
||||||
- has_keyword("internal")
|
- cog_find_peers
|
||||||
|
- cog_peer_compare
|
||||||
|
- cog_geographic_rollup
|
||||||
|
- title: Corpus metadata
|
||||||
|
desc: >
|
||||||
|
What the corpus contains, where a given result came from, and how to
|
||||||
|
hold a local copy of it.
|
||||||
|
contents:
|
||||||
|
- cog_categories
|
||||||
|
- cog_recipes
|
||||||
|
- cog_manifest
|
||||||
|
- cog_explain
|
||||||
|
- cog_mirror
|
||||||
|
|
||||||
articles:
|
articles:
|
||||||
- title: Getting started
|
- title: Concepts
|
||||||
navbar: ~
|
navbar: ~
|
||||||
contents: []
|
contents:
|
||||||
|
- total-spending
|
||||||
|
- population-denominators
|
||||||
|
|||||||
@@ -0,0 +1,238 @@
|
|||||||
|
# data-raw/regenerate_fixture_corpus.R
|
||||||
|
#
|
||||||
|
# Regenerate inst/extdata/fixture_corpus/ from a cog_pipeline publish tree.
|
||||||
|
#
|
||||||
|
# What this does:
|
||||||
|
# 1. Copies each requested year's long partition as-is (byte-for-byte)
|
||||||
|
# from <publish_cache>/data/long/ into the fixture. Default years are
|
||||||
|
# c(2011L, 2012L, 2019L, 2020L): 2011/2012 straddle the wide-aggregate
|
||||||
|
# -> modern-leaf format boundary (the harmonization/recipe seam), and
|
||||||
|
# 2019/2020 are the pre-existing per-capita/CPI regression anchors.
|
||||||
|
# Each partition is a full year (all states/govs) as published, so
|
||||||
|
# Broward County FL and every other previously-pinned government stay
|
||||||
|
# covered without any per-gov slicing logic.
|
||||||
|
# 2. Copies every metadata parquet the publish tree ships (see
|
||||||
|
# .FIXTURE_METADATA_FILES) as-is. These are small cross-vintage
|
||||||
|
# registries, not partitioned by year, so the fixture ships the complete
|
||||||
|
# tables rather than a year-scoped subset. representation.parquet and
|
||||||
|
# code_set.parquet are what make the sparse wide era interpretable --
|
||||||
|
# absence means "$0" in a dense_source year and "not reported" in a
|
||||||
|
# sparse_source one -- so a fixture without them cannot represent the
|
||||||
|
# published corpus.
|
||||||
|
# 3. Resyncs the four reference docs (data_dictionary.md,
|
||||||
|
# reader-specification.md, README.md, series_breaks.md) from the
|
||||||
|
# publish tree's docs/.
|
||||||
|
# 4. Hand-builds manifest.json for just the files the fixture ships,
|
||||||
|
# following the shape of the previous fixture manifest but with
|
||||||
|
# schema_version bumped to whatever the source manifest reports, and
|
||||||
|
# freshly computed sha256 / row_count / size_bytes for every fixture
|
||||||
|
# file (never copied from the source manifest, since paths and byte
|
||||||
|
# layout can differ subtly between a full corpus and a fixture).
|
||||||
|
#
|
||||||
|
# This is never a manual job: run it whenever cog_pipeline publishes a new
|
||||||
|
# corpus vintage that the fixture should track.
|
||||||
|
#
|
||||||
|
# Usage (from the uscogdata package root):
|
||||||
|
# Rscript data-raw/regenerate_fixture_corpus.R
|
||||||
|
# Rscript data-raw/regenerate_fixture_corpus.R /path/to/publish_cache
|
||||||
|
#
|
||||||
|
# Or from R:
|
||||||
|
# source("data-raw/regenerate_fixture_corpus.R")
|
||||||
|
# regenerate_fixture_corpus(publish_cache_dir = "/path/to/publish_cache")
|
||||||
|
|
||||||
|
# Every metadata parquet the publish tree ships, in the order they appear in
|
||||||
|
# the corpus manifest. Single source of truth for both the copy step and the
|
||||||
|
# fixture manifest, so the two can never drift apart.
|
||||||
|
.FIXTURE_METADATA_FILES <- c(
|
||||||
|
"canonical_alias.parquet",
|
||||||
|
"canonical_fips_xwalk.parquet",
|
||||||
|
"census_collection_coverage.parquet",
|
||||||
|
"code_set.parquet",
|
||||||
|
"harmonization_map.parquet",
|
||||||
|
"harmonization_recipes.parquet",
|
||||||
|
"lineage_events.parquet",
|
||||||
|
"representation.parquet",
|
||||||
|
"series_breaks.parquet",
|
||||||
|
"summary_categories.parquet"
|
||||||
|
)
|
||||||
|
|
||||||
|
regenerate_fixture_corpus <- function(
|
||||||
|
publish_cache_dir = file.path(
|
||||||
|
"..", "cog_pipeline", "_targets", "publish_cache"
|
||||||
|
),
|
||||||
|
fixture_dir = file.path("inst", "extdata", "fixture_corpus"),
|
||||||
|
fixture_years = c(2011L, 2012L, 2019L, 2020L)) {
|
||||||
|
stopifnot(
|
||||||
|
requireNamespace("digest", quietly = TRUE),
|
||||||
|
requireNamespace("jsonlite", quietly = TRUE),
|
||||||
|
requireNamespace("duckdb", quietly = TRUE),
|
||||||
|
requireNamespace("DBI", quietly = TRUE)
|
||||||
|
)
|
||||||
|
|
||||||
|
publish_cache_dir <- normalizePath(publish_cache_dir, mustWork = TRUE)
|
||||||
|
if (!dir.exists(fixture_dir)) dir.create(fixture_dir, recursive = TRUE)
|
||||||
|
|
||||||
|
source_manifest <- jsonlite::fromJSON(
|
||||||
|
file.path(publish_cache_dir, "manifest.json"),
|
||||||
|
simplifyVector = TRUE
|
||||||
|
)
|
||||||
|
|
||||||
|
.copy_long_partitions(publish_cache_dir, fixture_dir, fixture_years)
|
||||||
|
.copy_metadata_parquets(publish_cache_dir, fixture_dir)
|
||||||
|
.copy_docs(publish_cache_dir, fixture_dir)
|
||||||
|
|
||||||
|
manifest <- .build_fixture_manifest(
|
||||||
|
fixture_dir, source_manifest, fixture_years
|
||||||
|
)
|
||||||
|
manifest_path <- file.path(fixture_dir, "manifest.json")
|
||||||
|
writeLines(
|
||||||
|
jsonlite::toJSON(manifest, auto_unbox = TRUE, pretty = TRUE, null = "null"),
|
||||||
|
manifest_path
|
||||||
|
)
|
||||||
|
|
||||||
|
size_bytes <- sum(file.info(
|
||||||
|
list.files(fixture_dir, recursive = TRUE, full.names = TRUE)
|
||||||
|
)$size)
|
||||||
|
message(sprintf(
|
||||||
|
"Fixture corpus regenerated at %s (%.2f MB total).",
|
||||||
|
fixture_dir, size_bytes / 1024^2
|
||||||
|
))
|
||||||
|
invisible(manifest)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Copy each requested year's partition directory (just the parquet file
|
||||||
|
# inside it) from the publish tree into the fixture, as-is.
|
||||||
|
#' @noRd
|
||||||
|
.copy_long_partitions <- function(publish_cache_dir, fixture_dir, years) {
|
||||||
|
for (yr in years) {
|
||||||
|
part_rel <- file.path("data", "long", sprintf("year=%d", yr), "part-0.parquet")
|
||||||
|
src <- file.path(publish_cache_dir, part_rel)
|
||||||
|
dst <- file.path(fixture_dir, part_rel)
|
||||||
|
if (!file.exists(src)) {
|
||||||
|
stop(sprintf("Source partition missing: %s", src))
|
||||||
|
}
|
||||||
|
dir.create(dirname(dst), recursive = TRUE, showWarnings = FALSE)
|
||||||
|
ok <- file.copy(src, dst, overwrite = TRUE)
|
||||||
|
if (!ok) stop(sprintf("Failed to copy %s -> %s", src, dst))
|
||||||
|
}
|
||||||
|
invisible(NULL)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Copy the full (not year-scoped) metadata tables listed in
|
||||||
|
# .FIXTURE_METADATA_FILES.
|
||||||
|
#' @noRd
|
||||||
|
.copy_metadata_parquets <- function(publish_cache_dir, fixture_dir) {
|
||||||
|
for (f in .FIXTURE_METADATA_FILES) {
|
||||||
|
src <- file.path(publish_cache_dir, "data", f)
|
||||||
|
dst <- file.path(fixture_dir, "data", f)
|
||||||
|
if (!file.exists(src)) {
|
||||||
|
stop(sprintf("Source metadata file missing: %s", src))
|
||||||
|
}
|
||||||
|
dir.create(dirname(dst), recursive = TRUE, showWarnings = FALSE)
|
||||||
|
ok <- file.copy(src, dst, overwrite = TRUE)
|
||||||
|
if (!ok) stop(sprintf("Failed to copy %s -> %s", src, dst))
|
||||||
|
}
|
||||||
|
invisible(NULL)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Resync the four reference docs shipped alongside the fixture.
|
||||||
|
#' @noRd
|
||||||
|
.copy_docs <- function(publish_cache_dir, fixture_dir) {
|
||||||
|
docs <- c(
|
||||||
|
"data_dictionary.md", "reader-specification.md",
|
||||||
|
"README.md", "series_breaks.md"
|
||||||
|
)
|
||||||
|
dst_dir <- file.path(fixture_dir, "docs")
|
||||||
|
dir.create(dst_dir, recursive = TRUE, showWarnings = FALSE)
|
||||||
|
for (f in docs) {
|
||||||
|
src <- file.path(publish_cache_dir, "docs", f)
|
||||||
|
if (!file.exists(src)) {
|
||||||
|
stop(sprintf("Source doc missing: %s", src))
|
||||||
|
}
|
||||||
|
ok <- file.copy(src, file.path(dst_dir, f), overwrite = TRUE)
|
||||||
|
if (!ok) stop(sprintf("Failed to copy doc %s", f))
|
||||||
|
}
|
||||||
|
invisible(NULL)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Count rows in a parquet file via an ephemeral DuckDB connection.
|
||||||
|
#' @noRd
|
||||||
|
.parquet_row_count <- function(path) {
|
||||||
|
con <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE)
|
||||||
|
DBI::dbGetQuery(con, sprintf(
|
||||||
|
"SELECT COUNT(*) AS n FROM read_parquet(%s)",
|
||||||
|
.sql_quote(path)
|
||||||
|
))$n
|
||||||
|
}
|
||||||
|
|
||||||
|
#' @noRd
|
||||||
|
.sql_quote <- function(x) paste0("'", gsub("'", "''", x), "'")
|
||||||
|
|
||||||
|
# Hand-build manifest.json following the shape of the previous fixture
|
||||||
|
# manifest: schema_version / built_at / pipeline_commit / fixture_note /
|
||||||
|
# data_vintage / scope / schema / files.long_partitions / files.metadata /
|
||||||
|
# series_breaks_ref / reader_spec_ref. Every sha256 / row_count / size_bytes
|
||||||
|
# is freshly computed against the files actually written into fixture_dir.
|
||||||
|
#' @noRd
|
||||||
|
.build_fixture_manifest <- function(fixture_dir, source_manifest, years) {
|
||||||
|
long_partitions <- lapply(years, function(yr) {
|
||||||
|
rel <- file.path("data", "long", sprintf("year=%d", yr), "part-0.parquet")
|
||||||
|
path <- file.path(fixture_dir, rel)
|
||||||
|
list(
|
||||||
|
year = as.integer(yr),
|
||||||
|
path = gsub("\\\\", "/", rel),
|
||||||
|
sha256 = digest::digest(path, algo = "sha256", file = TRUE),
|
||||||
|
row_count = as.integer(.parquet_row_count(path)),
|
||||||
|
size_bytes = as.integer(file.info(path)$size)
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
metadata <- lapply(.FIXTURE_METADATA_FILES, function(f) {
|
||||||
|
rel <- file.path("data", f)
|
||||||
|
path <- file.path(fixture_dir, rel)
|
||||||
|
list(
|
||||||
|
path = gsub("\\\\", "/", rel),
|
||||||
|
sha256 = digest::digest(path, algo = "sha256", file = TRUE),
|
||||||
|
description = f
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
list(
|
||||||
|
schema_version = as.integer(source_manifest$schema_version),
|
||||||
|
built_at = format(Sys.time(), "%Y-%m-%dT%H:%M:%SZ", tz = "UTC"),
|
||||||
|
pipeline_commit = source_manifest$pipeline_commit,
|
||||||
|
fixture_note = paste(
|
||||||
|
"Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full",
|
||||||
|
"corpus available via USCOGDATA_URL. Regenerated from the sparsified",
|
||||||
|
"schema-v6 corpus: the wide era (<= FY2011) no longer stores explicit",
|
||||||
|
"zeros, so FY2011 absence means Census published $0 while FY2012+",
|
||||||
|
"absence means not reported. representation.parquet and",
|
||||||
|
"code_set.parquet carry that rule and ship in full, as do every other",
|
||||||
|
"metadata table in the publish tree. 2011/2012 straddle both the",
|
||||||
|
"wide-aggregate -> modern-leaf format boundary (exercised by",
|
||||||
|
"basis=\"harmonized\" and recipe= queries) and the dense -> sparse",
|
||||||
|
"representation boundary (SB194); 2019/2020 retain the prior",
|
||||||
|
"per-capita/CPI regression anchors. Regenerated via",
|
||||||
|
"data-raw/regenerate_fixture_corpus.R."
|
||||||
|
),
|
||||||
|
data_vintage = source_manifest$data_vintage,
|
||||||
|
scope = source_manifest$scope,
|
||||||
|
schema = source_manifest$schema,
|
||||||
|
files = list(
|
||||||
|
long_partitions = long_partitions,
|
||||||
|
metadata = metadata
|
||||||
|
),
|
||||||
|
series_breaks_ref = source_manifest$series_breaks_ref,
|
||||||
|
reader_spec_ref = source_manifest$reader_spec_ref
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
if (identical(environment(), globalenv()) && sys.nframe() == 0L) {
|
||||||
|
args <- commandArgs(trailingOnly = TRUE)
|
||||||
|
if (length(args) >= 1L) {
|
||||||
|
regenerate_fixture_corpus(publish_cache_dir = args[[1]])
|
||||||
|
} else {
|
||||||
|
regenerate_fixture_corpus()
|
||||||
|
}
|
||||||
|
}
|
||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+83
-49
@@ -1,82 +1,116 @@
|
|||||||
{
|
{
|
||||||
"schema_version": 3,
|
"schema_version": 6,
|
||||||
"built_at": "2026-04-27T16:43:46Z",
|
"built_at": "2026-08-03T16:51:32Z",
|
||||||
"pipeline_commit": "899af37",
|
"pipeline_commit": "e7394a4",
|
||||||
"fixture_note": "Two-year (2019-2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL.",
|
"fixture_note": "Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated from the sparsified schema-v6 corpus: the wide era (<= FY2011) no longer stores explicit zeros, so FY2011 absence means Census published $0 while FY2012+ absence means not reported. representation.parquet and code_set.parquet carry that rule and ship in full, as do every other metadata table in the publish tree. 2011/2012 straddle both the wide-aggregate -> modern-leaf format boundary (exercised by basis=\"harmonized\" and recipe= queries) and the dense -> sparse representation boundary (SB194); 2019/2020 retain the prior per-capita/CPI regression anchors. Regenerated via data-raw/regenerate_fixture_corpus.R.",
|
||||||
"data_vintage": {
|
"data_vintage": {
|
||||||
"census_source_downloaded": "unknown",
|
"source_vintages": {
|
||||||
"cpi_vintage": "FRED CPIAUCSL",
|
"2012": "10162019",
|
||||||
|
"2013": "10162019",
|
||||||
|
"2014": "10162019",
|
||||||
|
"2015": "10162019",
|
||||||
|
"2016": "10162019",
|
||||||
|
"2017": "06102021",
|
||||||
|
"2018": "06102021",
|
||||||
|
"2019": "06102021",
|
||||||
|
"2020": "06122023",
|
||||||
|
"2021": "06122023",
|
||||||
|
"2022": "06052025",
|
||||||
|
"2023": "06052025"
|
||||||
|
},
|
||||||
|
"registry_rows": 148,
|
||||||
"acs_vintage": "ACS 2018-2022 5-year"
|
"acs_vintage": "ACS 2018-2022 5-year"
|
||||||
},
|
},
|
||||||
"scope": {
|
"scope": {
|
||||||
"gov_types_included": [
|
"gov_types_included": [0, 1, 2, 3],
|
||||||
0,
|
"gov_types_excluded": [4, 5],
|
||||||
1,
|
|
||||||
2,
|
|
||||||
3
|
|
||||||
],
|
|
||||||
"gov_types_excluded": [
|
|
||||||
4,
|
|
||||||
5
|
|
||||||
],
|
|
||||||
"scope_note": "v0.1 covers state, county, city/municipality, and township governments. Special districts (type 4) and school districts (type 5) are excluded pending validation in a future cycle."
|
"scope_note": "v0.1 covers state, county, city/municipality, and township governments. Special districts (type 4) and school districts (type 5) are excluded pending validation in a future cycle."
|
||||||
},
|
},
|
||||||
"schema": {
|
"schema": {
|
||||||
"long_column_count": 24,
|
"long_column_count": 28,
|
||||||
"long_columns": [
|
"long_columns": ["fips_state", "type", "fips_county", "govid", "gov_blank", "gov_name", "county_name", "fips_state_asof", "fips_county_asof", "cog_legacy_state", "cog_legacy_county", "fips_place_code", "population", "popyear", "enrollment", "enrollyear", "function_code", "sch_level_code", "fiscal_year_end", "srvy_year", "item_code", "amt", "srv_data", "impute_flag", "is_aggregate", "canonical_govid", "harmonized_code", "survey_weight"],
|
||||||
"fips_state",
|
|
||||||
"type",
|
|
||||||
"fips_county",
|
|
||||||
"govid",
|
|
||||||
"gov_blank",
|
|
||||||
"gov_name",
|
|
||||||
"county_name",
|
|
||||||
"fips_state_code",
|
|
||||||
"fips_county_code",
|
|
||||||
"fips_place_code",
|
|
||||||
"population",
|
|
||||||
"popyear",
|
|
||||||
"enrollment",
|
|
||||||
"enrollyear",
|
|
||||||
"function_code",
|
|
||||||
"sch_level_code",
|
|
||||||
"fiscal_year_end",
|
|
||||||
"srvy_year",
|
|
||||||
"item_code",
|
|
||||||
"amt",
|
|
||||||
"srv_data",
|
|
||||||
"impute_flag",
|
|
||||||
"is_aggregate",
|
|
||||||
"canonical_govid"
|
|
||||||
],
|
|
||||||
"data_dictionary": "docs/data_dictionary.md"
|
"data_dictionary": "docs/data_dictionary.md"
|
||||||
},
|
},
|
||||||
"files": {
|
"files": {
|
||||||
"long_partitions": [
|
"long_partitions": [
|
||||||
|
{
|
||||||
|
"year": 2011,
|
||||||
|
"path": "data/long/year=2011/part-0.parquet",
|
||||||
|
"sha256": "7848e18497080c8980a4f89c5b386205b2c5bc90db6773827ea01ab3943d16b1",
|
||||||
|
"row_count": 496004,
|
||||||
|
"size_bytes": 2202455
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"year": 2012,
|
||||||
|
"path": "data/long/year=2012/part-0.parquet",
|
||||||
|
"sha256": "b82ac82d5e35f844b26c887445601f3748438c52c998ba4e403b025941a6f170",
|
||||||
|
"row_count": 1163338,
|
||||||
|
"size_bytes": 5929917
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"year": 2019,
|
"year": 2019,
|
||||||
"path": "data/long/year=2019/part-0.parquet",
|
"path": "data/long/year=2019/part-0.parquet",
|
||||||
"sha256": "e1c9f426c6d7d3c51d06b3a652473b987b304836619c213f019cee4887714daa",
|
"sha256": "5cbd4726dcc7d0dab5c2a05a64702e979533ae119ed0587073cd31c089e0d737",
|
||||||
"row_count": 318139,
|
"row_count": 318139,
|
||||||
"size_bytes": 1424231
|
"size_bytes": 1719548
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"year": 2020,
|
"year": 2020,
|
||||||
"path": "data/long/year=2020/part-0.parquet",
|
"path": "data/long/year=2020/part-0.parquet",
|
||||||
"sha256": "9b795853a848e8c955c80261b96b79630fc77394dcfb1a1ca288e2cd634053a3",
|
"sha256": "ee548fec80bf1beda844fe03916ac145f10dd34c45968407cc330ec260935f00",
|
||||||
"row_count": 317500,
|
"row_count": 317500,
|
||||||
"size_bytes": 1427150
|
"size_bytes": 1722918
|
||||||
}
|
}
|
||||||
],
|
],
|
||||||
"metadata": [
|
"metadata": [
|
||||||
|
{
|
||||||
|
"path": "data/canonical_alias.parquet",
|
||||||
|
"sha256": "3f617051c23a99bea322889857f7106df0c92954564afeec181df7083ee6698e",
|
||||||
|
"description": "canonical_alias.parquet"
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"path": "data/canonical_fips_xwalk.parquet",
|
"path": "data/canonical_fips_xwalk.parquet",
|
||||||
"sha256": "86e53e04a35f6f90bb74bb1a273e053392afa782d6f518e3e3da9c976d47f7af",
|
"sha256": "f98742f941269dacf8f7de5c273aa4dd4e75017a5bb70c054da35852a95a8d46",
|
||||||
"description": "canonical_fips_xwalk.parquet"
|
"description": "canonical_fips_xwalk.parquet"
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
"path": "data/census_collection_coverage.parquet",
|
||||||
|
"sha256": "143e025616cde684da7c4442bc00d07fbd1556fabb0ea96223931b737e5d10a4",
|
||||||
|
"description": "census_collection_coverage.parquet"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"path": "data/code_set.parquet",
|
||||||
|
"sha256": "4cffcb0198dd51e4ff2b694050bb371a5f9965cdac12f25521cb628fb8e118a9",
|
||||||
|
"description": "code_set.parquet"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"path": "data/harmonization_map.parquet",
|
||||||
|
"sha256": "4cf32d0f817079ba4f28dc0ce65450d3247ebbf08d94c0c26c0d02af597bf812",
|
||||||
|
"description": "harmonization_map.parquet"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"path": "data/harmonization_recipes.parquet",
|
||||||
|
"sha256": "1133e9a0b02f8f34f5f936e55c5ecd596bb8a55d8425dcce76767f0f3203581c",
|
||||||
|
"description": "harmonization_recipes.parquet"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"path": "data/lineage_events.parquet",
|
||||||
|
"sha256": "36c16acfbe621d61010984767f1c566993b8a5f481a2c1e134c4c0a600e4502f",
|
||||||
|
"description": "lineage_events.parquet"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"path": "data/representation.parquet",
|
||||||
|
"sha256": "31ec328a7dd505a321b45f97aafff12e53d68a1a986f63509863035b22a4360d",
|
||||||
|
"description": "representation.parquet"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"path": "data/series_breaks.parquet",
|
||||||
|
"sha256": "731998516cd802f63fcf7fb66053c7a62b7be955ab0794cad4a4979cb7628b87",
|
||||||
|
"description": "series_breaks.parquet"
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"path": "data/summary_categories.parquet",
|
"path": "data/summary_categories.parquet",
|
||||||
"sha256": "60045e22bc2723318fa2cb73f8e5038250dc54d24b3447c6750dfe29035335b8",
|
"sha256": "e3b0efa00ce713b8f45829b89cfde24b55333f26101f0495df82d85997d18d8e",
|
||||||
"description": "summary_categories.parquet"
|
"description": "summary_categories.parquet"
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
|
|||||||
@@ -10,11 +10,100 @@
|
|||||||
"target": { "type": "object" },
|
"target": { "type": "object" },
|
||||||
"years": { "type": "array", "items": { "type": "integer" } },
|
"years": { "type": "array", "items": { "type": "integer" } },
|
||||||
"category": { "type": ["string", "array", "null"] },
|
"category": { "type": ["string", "array", "null"] },
|
||||||
|
"basis": { "type": ["string", "null"] },
|
||||||
|
"basis_note": { "type": ["string", "null"] },
|
||||||
|
"expenditure_concept": {
|
||||||
|
"type": "string",
|
||||||
|
"enum": ["primary", "direct", "total"],
|
||||||
|
"description": "Which spending concept produced this result, defined as crosswalk spend_subtype sets (never item-code prefixes). 'primary' (the default) is the government's own service provision: operations + capital + assistance. 'direct' adds interest on debt and insurance trust benefit payments (Census's published Direct Expenditure). 'total' adds intergovernmental payments (M to local governments, L to state government, Q11/Q12/Q18 to school systems). Only 'primary' and 'direct' are valid for results combined across governments."
|
||||||
|
},
|
||||||
|
"expenditure_concept_note": {
|
||||||
|
"type": ["string", "null"],
|
||||||
|
"description": "How the intergovernmental leg was assembled; null for 'primary' and 'direct'."
|
||||||
|
},
|
||||||
|
"expenditure_concept_direct_suppressed": {
|
||||||
|
"type": ["boolean", "null"],
|
||||||
|
"description": "TRUE when expenditure_concept = 'total' and at least one requested (year, category) has intergovernmental rows but NO Direct rows in this corpus (typically a legacy aggregate-only family) -- those result rows report the intergovernmental leg alone, not Direct + IG. Always FALSE for expenditure_concept = 'primary' or 'direct'. null (NA) when expenditure_concept = 'total' AND category = 'All Categories': the detector keys on per-category rows, which that mode collapses, so suppression cannot be computed -- see `expenditure_concept_note`. See the affected rows' `notes` for the recovering recipe, if any."
|
||||||
|
},
|
||||||
|
"revenue_concept": {
|
||||||
|
"type": "string",
|
||||||
|
"enum": ["general", "total"],
|
||||||
|
"description": "Which revenue concept produced this result, defined as crosswalk revenue_subtype sets (never item-code prefixes). 'general' (the default) is Census General Revenue: own_source + federal + state + local_aid. 'total' is Census Total Revenue: general plus utility, liquor store and insurance trust revenue. Census defines the first by subtracting the other three from the second (manual section 4.3). Meaningful for cog_revenue() results; spending results carry the default.",
|
||||||
|
"$comment": "The employee-retirement (X) codes inside insurance_trust stop at FY2016, so a 'total' series steps at the FY2016/FY2017 seam for collection-scope reasons (series breaks SB197-SB202)."
|
||||||
|
},
|
||||||
|
"harmonization": { "type": "object" },
|
||||||
|
"recipe": { "type": ["object", "null"] },
|
||||||
|
"suggestions": {
|
||||||
|
"type": "array",
|
||||||
|
"description": "Harmonization recipes that would fill incomplete coverage in the requested years for this government. Empty on a healthy query, on an un-scoped (category = NULL) query, on basis = 'raw', and on a recipe = query (which resolves its own coverage).",
|
||||||
|
"items": {
|
||||||
|
"type": "object",
|
||||||
|
"required": ["recipe_id", "label", "available_years", "hint", "ig_recipe_id",
|
||||||
|
"trigger", "suppressed_amount", "suppressed_years", "suppressed_codes"],
|
||||||
|
"properties": {
|
||||||
|
"recipe_id": { "type": "string" },
|
||||||
|
"label": { "type": "string" },
|
||||||
|
"available_years": {
|
||||||
|
"type": "array",
|
||||||
|
"items": { "type": "integer" },
|
||||||
|
"description": "[year_min, year_max] of the recipe's component coverage."
|
||||||
|
},
|
||||||
|
"hint": { "type": "string" },
|
||||||
|
"ig_recipe_id": {
|
||||||
|
"type": ["string", "null"],
|
||||||
|
"description": "The intergovernmental (M/L) counterpart recipe covering the same function suffixes, or null. Never set for revenue recipes."
|
||||||
|
},
|
||||||
|
"trigger": {
|
||||||
|
"type": "string",
|
||||||
|
"enum": ["empty_year", "suppressed_component"],
|
||||||
|
"description": "Why this fired. 'empty_year': the result has no rows at all in a requested year. 'suppressed_component': the result HAS rows, but a component code carries dollars this government reports in the requested years that the verb's underlying long view structurally excludes -- aggregate-published, carrying no harmonized code, or absent from summary_categories. This is NOT the same thing as 'excluded from the result': a component present in the view under a different category (a scoping choice, e.g. a different `category` or a narrower `expenditure_concept`) contributes 0 and never fires. 'empty_year' wins when both apply, being the stronger claim; the suppressed_* fields are populated either way, using the same underlying-view measurement, and can be 0 even on an 'empty_year' fire."
|
||||||
|
},
|
||||||
|
"suppressed_amount": {
|
||||||
|
"type": "number",
|
||||||
|
"description": "Full US dollars this government reports, in the recipe's component codes, in the requested years, that the verb's underlying long view structurally excludes (aggregate-published, carrying no harmonized code, or absent from summary_categories) -- summed across those years. This is NOT the same quantity as 'what the result excludes': a component present in the view under a different category or a narrower `expenditure_concept` is scoped out on purpose, counts as 0 here, and is not suppression. 0 does not always mean full coverage -- see 'trigger' and 'empty_year'. May be negative where Census publishes a negative `amt` for the excluded rows."
|
||||||
|
},
|
||||||
|
"suppressed_years": {
|
||||||
|
"type": "array",
|
||||||
|
"items": { "type": "integer" },
|
||||||
|
"description": "The requested years contributing to suppressed_amount."
|
||||||
|
},
|
||||||
|
"suppressed_codes": {
|
||||||
|
"type": "array",
|
||||||
|
"items": { "type": "string" },
|
||||||
|
"description": "The excluded component item codes, sorted."
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
"scope": { "type": "object" },
|
"scope": { "type": "object" },
|
||||||
"codes_summed": { "type": "object" },
|
"codes_summed": { "type": "object" },
|
||||||
"aggregate_fallback": { "type": ["object", "null"] },
|
"aggregate_fallback": { "type": ["object", "null"] },
|
||||||
"transformations":{ "type": "object" },
|
"transformations":{ "type": "object" },
|
||||||
"series_break_refs": { "type": "array", "items": { "type": "string" } },
|
"series_break_refs": { "type": "array", "items": { "type": "string" } },
|
||||||
|
"completion": {
|
||||||
|
"type": "object",
|
||||||
|
"description": "What `complete = TRUE` filled. `applied` is FALSE on an ordinary query. `rows_filled` counts cells added to the requested grid, and `absence_means` maps each requested year to the meaning of an absent cell there ('census_zero' in a dense_source year, 'not_reported' in a sparse_source one). Filled rows carry `value_source` in the result: 'reported', 'census_zero' (amount 0 -- Census published $0), or 'not_reported' (amount NA -- unknown).",
|
||||||
|
"properties": {
|
||||||
|
"applied": { "type": "boolean" },
|
||||||
|
"rows_filled": { "type": "integer" },
|
||||||
|
"absence_means": { "type": "object" }
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"corpus_break_refs": {
|
||||||
|
"type": "array",
|
||||||
|
"items": { "type": "string" },
|
||||||
|
"description": "Ids of catalogued series breaks whose fin_code is the literal 'ALL' -- caveats about the corpus as a whole (dollar precision across 1976/1977, imputation exclusion from 2002, the dense -> sparse representation change at 2012, the government id scheme change at 2017) rather than about one item code. Selected on the break_year window alone, so they do not depend on which codes a result contains. Disjoint from series_break_refs by construction: an entry qualifies the whole result, not one series."
|
||||||
|
},
|
||||||
|
"balance_caveats": {
|
||||||
|
"type": ["object", "null"],
|
||||||
|
"description": "Present only on cog_balances() results (null/absent for cog_spending()/cog_revenue()). `not_gaap` is always TRUE and `not_gaap_note` explains that Census holdings are gross -- no liabilities are netted -- so they are NOT comparable to a GAAP fund balance. `coverage_window` maps EVERY balance_subtype present in the mounted corpus -- not only the ones this query observed -- to its measured [min year, max year] there (never hardcoded), so a caller can see which families exist and over what span before deciding they missed one. `truncated` is the query-scoped field: it lists only the subtypes this result actually observed whose coverage_window does not fully span the requested years.",
|
||||||
|
"properties": {
|
||||||
|
"not_gaap": { "type": "boolean" },
|
||||||
|
"not_gaap_note": { "type": "string" },
|
||||||
|
"coverage_window": { "type": "object" },
|
||||||
|
"truncated": { "type": "array", "items": { "type": "string" } }
|
||||||
|
}
|
||||||
|
},
|
||||||
"manifest": { "type": "object" },
|
"manifest": { "type": "object" },
|
||||||
"sql_query": { "type": "string" }
|
"sql_query": { "type": "string" }
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,3 +1,7 @@
|
|||||||
CREATE OR REPLACE VIEW long AS
|
CREATE OR REPLACE VIEW long AS
|
||||||
SELECT *
|
SELECT *
|
||||||
FROM read_parquet('{url}data/long/**/*.parquet', hive_partitioning = true);
|
-- {long_files} carries its own quoting: a bracketed list of every partition
|
||||||
|
-- the manifest enumerates, or a single quoted glob on fallback. Do NOT wrap
|
||||||
|
-- it in quotes. See .long_files_sql() in R/views.R for why a glob alone
|
||||||
|
-- cannot work over HTTP.
|
||||||
|
FROM read_parquet({long_files}, hive_partitioning = true);
|
||||||
|
|||||||
@@ -0,0 +1,7 @@
|
|||||||
|
-- Category crosswalk. Numbered 11 (not with the other reference tables at
|
||||||
|
-- 30+) because the flow views (20-25) classify by MEMBERSHIP in this table
|
||||||
|
-- and DuckDB binds a view's sources eagerly at CREATE VIEW time, so it must
|
||||||
|
-- already exist when they register.
|
||||||
|
CREATE OR REPLACE VIEW summary_categories AS
|
||||||
|
SELECT *
|
||||||
|
FROM read_parquet('{url}data/summary_categories.parquet');
|
||||||
@@ -1,5 +1,22 @@
|
|||||||
|
-- Direct-side expenditure rows, classified by crosswalk MEMBERSHIP
|
||||||
|
-- (summary_categories.category_type = 'expenditure'), never by item-code
|
||||||
|
-- first letter: prefix Y alone spans revenue (Y01/Y02), expenditure
|
||||||
|
-- (Y05/Y06) and balance codes, so no first-letter allowlist can route it
|
||||||
|
-- (uscogdata#11, finding F-018). Which subtypes a query actually returns is
|
||||||
|
-- decided per expenditure_concept in R (.verb_spendrev); this view carries
|
||||||
|
-- every non-intergovernmental expenditure subtype: operations, capital,
|
||||||
|
-- assistance, interest, insurance_benefits.
|
||||||
|
--
|
||||||
|
-- The intergovernmental subtype (M/L/Q codes) is deliberately carved out
|
||||||
|
-- into ig_long: its legacy-era rows are published ONLY as aggregate-flagged
|
||||||
|
-- rows, so it cannot live behind this view's NOT is_aggregate filter (see
|
||||||
|
-- 24-ig_long.sql).
|
||||||
CREATE OR REPLACE VIEW spending_long AS
|
CREATE OR REPLACE VIEW spending_long AS
|
||||||
SELECT *
|
SELECT *
|
||||||
FROM long
|
FROM long
|
||||||
WHERE LEFT(item_code, 1) IN ('E', 'F', 'G', 'K')
|
WHERE item_code IN (
|
||||||
|
SELECT item_code FROM summary_categories
|
||||||
|
WHERE category_type = 'expenditure'
|
||||||
|
AND spend_subtype <> 'intergovernmental'
|
||||||
|
)
|
||||||
AND NOT is_aggregate;
|
AND NOT is_aggregate;
|
||||||
|
|||||||
@@ -1,5 +1,18 @@
|
|||||||
|
-- Revenue rows, classified by crosswalk MEMBERSHIP rather than item-code
|
||||||
|
-- first letter (see 20-spending_long.sql for why prefixes cannot work).
|
||||||
|
--
|
||||||
|
-- Carries EVERY revenue subtype. Which of Census's two published concepts a
|
||||||
|
-- query actually returns is decided per revenue_concept in R
|
||||||
|
-- (.verb_spendrev), exactly as expenditure_concept narrows spending_long:
|
||||||
|
-- general = own_source + federal + state + local_aid (the default)
|
||||||
|
-- total = general + utility + liquor_store + insurance_trust
|
||||||
|
-- Census defines the first by subtracting the other three from the second
|
||||||
|
-- (manual section 4.3), so both concepts need all four families present here.
|
||||||
CREATE OR REPLACE VIEW revenue_long AS
|
CREATE OR REPLACE VIEW revenue_long AS
|
||||||
SELECT *
|
SELECT *
|
||||||
FROM long
|
FROM long
|
||||||
WHERE LEFT(item_code, 1) IN ('T', 'A', 'U', 'B', 'C', 'D')
|
WHERE item_code IN (
|
||||||
|
SELECT item_code FROM summary_categories
|
||||||
|
WHERE category_type = 'revenue'
|
||||||
|
)
|
||||||
AND NOT is_aggregate;
|
AND NOT is_aggregate;
|
||||||
|
|||||||
@@ -0,0 +1,16 @@
|
|||||||
|
-- Harmonized-basis twin of 20-spending_long.sql: same crosswalk-membership
|
||||||
|
-- classification, applied to harmonized_code (the code the row is folded
|
||||||
|
-- onto) rather than the published item_code. Safe because the harmonized
|
||||||
|
-- space is leaf-only and every harmonized_code in the corpus is a
|
||||||
|
-- summary_categories member (verified at fixture regen; a code the
|
||||||
|
-- crosswalk cannot classify would be silently dropped here).
|
||||||
|
CREATE OR REPLACE VIEW spending_long_harmonized AS
|
||||||
|
SELECT * REPLACE (harmonized_code AS item_code)
|
||||||
|
FROM long
|
||||||
|
WHERE NOT is_aggregate
|
||||||
|
AND harmonized_code IS NOT NULL
|
||||||
|
AND harmonized_code IN (
|
||||||
|
SELECT item_code FROM summary_categories
|
||||||
|
WHERE category_type = 'expenditure'
|
||||||
|
AND spend_subtype <> 'intergovernmental'
|
||||||
|
);
|
||||||
@@ -0,0 +1,12 @@
|
|||||||
|
-- Harmonized-basis twin of 21-revenue_long.sql: same crosswalk-membership
|
||||||
|
-- classification (every revenue subtype; the concept narrows in R), applied
|
||||||
|
-- to harmonized_code rather than the published item_code.
|
||||||
|
CREATE OR REPLACE VIEW revenue_long_harmonized AS
|
||||||
|
SELECT * REPLACE (harmonized_code AS item_code)
|
||||||
|
FROM long
|
||||||
|
WHERE NOT is_aggregate
|
||||||
|
AND harmonized_code IS NOT NULL
|
||||||
|
AND harmonized_code IN (
|
||||||
|
SELECT item_code FROM summary_categories
|
||||||
|
WHERE category_type = 'revenue'
|
||||||
|
);
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
-- Intergovernmental expenditure rows: crosswalk spend_subtype =
|
||||||
|
-- 'intergovernmental' (M = to local govts, L = to state govts, Q11/Q12/Q18
|
||||||
|
-- = state payments to school systems -- uscogdata#11, finding F-017).
|
||||||
|
--
|
||||||
|
-- Deliberately does NOT filter `NOT is_aggregate`, unlike spending_long. In the
|
||||||
|
-- wide era (<= FY2011) the IG families M05/M12/M47/M89/L47/L89 are published
|
||||||
|
-- ONLY as aggregate-flagged rows -- filtering them would hide ~70% of legacy IG
|
||||||
|
-- dollars and make Total silently collapse to Direct. This is safe because the
|
||||||
|
-- aggregate codes and their modern leaf components are strictly year-disjoint
|
||||||
|
-- (M47 ends 2011 / M94 starts 2012; M89 is aggregate only <= 2011 and a leaf
|
||||||
|
-- from 2012 alongside M91-93), so no row is ever counted twice. Same argument
|
||||||
|
-- the pipeline's recipe joins use.
|
||||||
|
--
|
||||||
|
-- `L--` stays excluded: it is the IG-to-state FAMILY TOTAL and genuinely
|
||||||
|
-- rolls up the L-NN codes, so including it would double-count. The crosswalk
|
||||||
|
-- deliberately carries no `--` family-total codes, so membership excludes it
|
||||||
|
-- (guarded by "the IG leg never includes the L-- family total" in
|
||||||
|
-- tests/testthat/test-expenditure-concept.R).
|
||||||
|
CREATE OR REPLACE VIEW ig_long AS
|
||||||
|
SELECT *
|
||||||
|
FROM long
|
||||||
|
WHERE item_code IN (
|
||||||
|
SELECT item_code FROM summary_categories
|
||||||
|
WHERE spend_subtype = 'intergovernmental'
|
||||||
|
);
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
-- Harmonized-basis IG rows. Uses COALESCE(harmonized_code, item_code) rather
|
||||||
|
-- than harmonized_code alone: aggregate rows carry NO harmonized_code by
|
||||||
|
-- construction (harmonized space is leaf-only), so a plain
|
||||||
|
-- `harmonized_code IS NOT NULL` filter would drop every legacy IG aggregate --
|
||||||
|
-- in the bundled fixture corpus (year 2011; 2012+ all carry a harmonized_code)
|
||||||
|
-- that is $379,016,063k across 25,688 M rows and $2,277,458k across 19,266 L
|
||||||
|
-- rows (`SELECT year, LEFT(item_code,1), SUM(amt), COUNT(*) FROM ig_long
|
||||||
|
-- WHERE harmonized_code IS NULL GROUP BY 1, 2`). COALESCE keeps the one real
|
||||||
|
-- IG collapse rule (M38 -> M36, SB012, year-disjoint 1967-2011 vs 2012+)
|
||||||
|
-- while never dropping a row.
|
||||||
|
--
|
||||||
|
-- Membership is checked on the published item_code (mirroring 24-ig_long.sql)
|
||||||
|
-- rather than the COALESCEd code: every IG harmonization target (M36) is
|
||||||
|
-- itself an IG crosswalk member, so the two are equivalent, and item_code is
|
||||||
|
-- the column that exists on every row.
|
||||||
|
CREATE OR REPLACE VIEW ig_long_harmonized AS
|
||||||
|
SELECT * REPLACE (COALESCE(harmonized_code, item_code) AS item_code)
|
||||||
|
FROM long
|
||||||
|
WHERE item_code IN (
|
||||||
|
SELECT item_code FROM summary_categories
|
||||||
|
WHERE spend_subtype = 'intergovernmental'
|
||||||
|
);
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
-- Cash and security holdings, classified by crosswalk MEMBERSHIP on
|
||||||
|
-- category_type (see 21-revenue_long.sql for why first-letter prefixes cannot
|
||||||
|
-- do this job -- the X and Y families each span revenue, expenditure AND
|
||||||
|
-- balance).
|
||||||
|
--
|
||||||
|
-- These rows are STOCKS: a balance at a point in time, not a flow over a
|
||||||
|
-- fiscal year. Summing a stock with a flow is meaningless, which is why they
|
||||||
|
-- live behind a third view rather than as a subtype of either money view, and
|
||||||
|
-- why neither spending_long nor revenue_long can reach them.
|
||||||
|
--
|
||||||
|
-- `NOT is_aggregate` mirrors spending_long / revenue_long. The wide-era
|
||||||
|
-- aggregate-only holdings codes (X40/X41) are deliberately outside this view;
|
||||||
|
-- they are reachable only through the recipe path, which bypasses this filter
|
||||||
|
-- by design (cog_pipeline/docs/phase_r_harmonization_review.md § 0.2).
|
||||||
|
CREATE OR REPLACE VIEW balance_long AS
|
||||||
|
SELECT *
|
||||||
|
FROM long
|
||||||
|
WHERE item_code IN (
|
||||||
|
SELECT item_code FROM summary_categories
|
||||||
|
WHERE category_type = 'balance'
|
||||||
|
)
|
||||||
|
AND NOT is_aggregate;
|
||||||
@@ -1,3 +0,0 @@
|
|||||||
CREATE OR REPLACE VIEW summary_categories AS
|
|
||||||
SELECT *
|
|
||||||
FROM read_parquet('{url}data/summary_categories.parquet');
|
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
CREATE OR REPLACE VIEW gov_population_yearly AS
|
||||||
|
SELECT DISTINCT
|
||||||
|
year,
|
||||||
|
canonical_govid,
|
||||||
|
population,
|
||||||
|
popyear
|
||||||
|
FROM long
|
||||||
|
WHERE population IS NOT NULL;
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
CREATE OR REPLACE VIEW harmonization_map AS
|
||||||
|
SELECT *
|
||||||
|
FROM read_parquet('{url}data/harmonization_map.parquet');
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
CREATE OR REPLACE VIEW harmonization_recipes AS
|
||||||
|
SELECT *
|
||||||
|
FROM read_parquet('{url}data/harmonization_recipes.parquet');
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
CREATE OR REPLACE VIEW series_breaks_pq AS
|
||||||
|
SELECT *
|
||||||
|
FROM read_parquet('{url}data/series_breaks.parquet');
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
CREATE OR REPLACE VIEW representation AS
|
||||||
|
SELECT *
|
||||||
|
FROM read_parquet('{url}data/representation.parquet');
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
CREATE OR REPLACE VIEW code_set AS
|
||||||
|
SELECT *
|
||||||
|
FROM read_parquet('{url}data/code_set.parquet');
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
CREATE OR REPLACE VIEW spending_annotated_harmonized AS
|
||||||
|
SELECT
|
||||||
|
s.*,
|
||||||
|
x.gov_name AS xwalk_gov_name,
|
||||||
|
x.govs_type,
|
||||||
|
x.type_label,
|
||||||
|
x.fips_state AS xwalk_fips_state,
|
||||||
|
x.fips_county AS xwalk_fips_county,
|
||||||
|
x.fips_place,
|
||||||
|
x.population_acs,
|
||||||
|
c.category,
|
||||||
|
c.category_type,
|
||||||
|
c.spend_subtype
|
||||||
|
FROM spending_long_harmonized s
|
||||||
|
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
|
||||||
|
LEFT JOIN summary_categories c USING (item_code);
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
CREATE OR REPLACE VIEW revenue_annotated_harmonized AS
|
||||||
|
SELECT
|
||||||
|
s.*,
|
||||||
|
x.gov_name AS xwalk_gov_name,
|
||||||
|
x.govs_type,
|
||||||
|
x.type_label,
|
||||||
|
x.fips_state AS xwalk_fips_state,
|
||||||
|
x.fips_county AS xwalk_fips_county,
|
||||||
|
x.fips_place,
|
||||||
|
x.population_acs,
|
||||||
|
c.category,
|
||||||
|
c.category_type,
|
||||||
|
c.revenue_subtype
|
||||||
|
FROM revenue_long_harmonized s
|
||||||
|
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
|
||||||
|
LEFT JOIN summary_categories c USING (item_code);
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
CREATE OR REPLACE VIEW ig_annotated AS
|
||||||
|
SELECT
|
||||||
|
s.*,
|
||||||
|
x.gov_name AS xwalk_gov_name,
|
||||||
|
x.govs_type,
|
||||||
|
x.type_label,
|
||||||
|
x.fips_state AS xwalk_fips_state,
|
||||||
|
x.fips_county AS xwalk_fips_county,
|
||||||
|
x.fips_place,
|
||||||
|
x.population_acs,
|
||||||
|
c.category,
|
||||||
|
c.category_type,
|
||||||
|
c.spend_subtype
|
||||||
|
FROM ig_long s
|
||||||
|
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
|
||||||
|
LEFT JOIN summary_categories c USING (item_code);
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
CREATE OR REPLACE VIEW ig_annotated_harmonized AS
|
||||||
|
SELECT
|
||||||
|
s.*,
|
||||||
|
x.gov_name AS xwalk_gov_name,
|
||||||
|
x.govs_type,
|
||||||
|
x.type_label,
|
||||||
|
x.fips_state AS xwalk_fips_state,
|
||||||
|
x.fips_county AS xwalk_fips_county,
|
||||||
|
x.fips_place,
|
||||||
|
x.population_acs,
|
||||||
|
c.category,
|
||||||
|
c.category_type,
|
||||||
|
c.spend_subtype
|
||||||
|
FROM ig_long_harmonized s
|
||||||
|
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
|
||||||
|
LEFT JOIN summary_categories c USING (item_code);
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
CREATE OR REPLACE VIEW balance_annotated AS
|
||||||
|
SELECT
|
||||||
|
s.*,
|
||||||
|
x.gov_name AS xwalk_gov_name,
|
||||||
|
x.govs_type,
|
||||||
|
x.type_label,
|
||||||
|
x.fips_state AS xwalk_fips_state,
|
||||||
|
x.fips_county AS xwalk_fips_county,
|
||||||
|
x.fips_place,
|
||||||
|
x.population_acs,
|
||||||
|
c.category,
|
||||||
|
c.category_type,
|
||||||
|
c.balance_subtype
|
||||||
|
FROM balance_long s
|
||||||
|
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
|
||||||
|
LEFT JOIN summary_categories c USING (item_code);
|
||||||
@@ -0,0 +1,108 @@
|
|||||||
|
% Generated by roxygen2: do not edit by hand
|
||||||
|
% Please edit documentation in R/balances.R
|
||||||
|
\name{cog_balances}
|
||||||
|
\alias{cog_balances}
|
||||||
|
\title{Cash and security holdings for one or more governments}
|
||||||
|
\usage{
|
||||||
|
cog_balances(
|
||||||
|
govid = NULL,
|
||||||
|
years,
|
||||||
|
category = NULL,
|
||||||
|
per_capita = FALSE,
|
||||||
|
adjust_to_year = NULL,
|
||||||
|
basis = c("harmonized", "raw"),
|
||||||
|
recipe = NULL,
|
||||||
|
state = NULL,
|
||||||
|
type = NULL
|
||||||
|
)
|
||||||
|
}
|
||||||
|
\arguments{
|
||||||
|
\item{govid}{Canonical govid(s): a character vector, or a data frame with a
|
||||||
|
`canonical_govid` column (e.g. from [cog_gov_search()]). `NULL` to name
|
||||||
|
the cohort by `state`/`type` instead.}
|
||||||
|
|
||||||
|
\item{years}{Integer vector of fiscal years.}
|
||||||
|
|
||||||
|
\item{category}{Optional character vector of categories to keep. One of
|
||||||
|
`"Fund Balances"`, `"Insurance Trust Balances"`,
|
||||||
|
`"Retirement System Holdings"`. There is deliberately no `subtype`
|
||||||
|
argument: for holdings, `category` is a strict coarsening of
|
||||||
|
`balance_subtype` (unlike the money verbs, where the two axes cross), so
|
||||||
|
every combination would be either redundant or empty.
|
||||||
|
`category = "Fund Balances"` is exactly the `general` family
|
||||||
|
(`W01`/`W31`/`W61`). `balance_subtype` is returned, so a finer split is
|
||||||
|
one `dplyr::filter()` away. The reserved pseudo-category
|
||||||
|
`"All Categories"` (see [cog_spending()]) is **not** supported here and
|
||||||
|
errors with class `uscogdata_all_categories_unsupported`: it sums a
|
||||||
|
concept's subtype scope, and holdings are a stock with no concept
|
||||||
|
vocabulary to sum across. Omit `category` to get every category broken
|
||||||
|
out instead.}
|
||||||
|
|
||||||
|
\item{per_capita}{Divide holdings by population. Note this is a **stock per
|
||||||
|
resident** (reserves per person), which is *not* comparable to
|
||||||
|
[cog_spending()]'s per-capita figures -- those are a flow per person.}
|
||||||
|
|
||||||
|
\item{adjust_to_year}{Deflate to this year's dollars (CPI-U).}
|
||||||
|
|
||||||
|
\item{basis}{Accepted for uniformity with the money verbs, but currently a
|
||||||
|
**no-op**: `harmonization_map` carries no balance-code rows, so harmonized
|
||||||
|
and raw space are identical for holdings. Reported in
|
||||||
|
`provenance$basis_note`.}
|
||||||
|
|
||||||
|
\item{recipe}{Optional harmonization recipe id (see [cog_recipes()]).
|
||||||
|
`"cash_securities_z77_wide"` and `"cash_securities_z78_wide"` bridge the
|
||||||
|
wide era to the modern one.}
|
||||||
|
|
||||||
|
\item{state, type}{Name the cohort by predicate instead of by id: `state` is
|
||||||
|
a 2-letter USPS abbreviation (or a FIPS code) and `type` is one of
|
||||||
|
`"state"`, `"county"`, `"city"`, `"township"` (or the integer `0:3`) --
|
||||||
|
the same vocabulary, and the same internal coercion, as
|
||||||
|
[cog_gov_search()]. Both default to `NULL`.
|
||||||
|
|
||||||
|
The cohort is then expressed as a subquery against `canonical_fips_xwalk`
|
||||||
|
inside each statement rather than round-tripped through R as a literal id
|
||||||
|
list. For a fleet-scale cohort that is the difference between a
|
||||||
|
301,591-character `IN` list re-parsed in 5--8 statements per call and a
|
||||||
|
constant-size predicate: measured at **94 ms versus 449 ms** for the same
|
||||||
|
FY2022 aggregate over the 20,106-government `type = "city"` cohort, within
|
||||||
|
7% of the no-filter floor.
|
||||||
|
|
||||||
|
Supplying `govid` **and** `state`/`type` INTERSECTS them -- the
|
||||||
|
governments in `govid` that also match the predicate -- rather than one
|
||||||
|
silently taking precedence. Naming no cohort at all (`govid`, `state` and
|
||||||
|
`type` all `NULL`) aborts with class `uscogdata_no_cohort`.
|
||||||
|
|
||||||
|
When the cohort is named by predicate, `provenance$scope$govids_found`
|
||||||
|
and `govids_missing` are empty -- there is no id list to report against --
|
||||||
|
and `provenance$scope$cohort` carries `state`, `type` and
|
||||||
|
`n_governments` instead. A `govid`-named cohort reports exactly as before.}
|
||||||
|
}
|
||||||
|
\value{
|
||||||
|
Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||||
|
`balance_subtype`, `category`, `amt_nominal`, `codes_included`,
|
||||||
|
`aggregate_fallback`, plus optional `amt_per_capita_nominal` and
|
||||||
|
`pop_source` (when `per_capita = TRUE`), optional `amt_real` (when
|
||||||
|
`adjust_to_year` is set), and optional `amt_per_capita_real` (only when
|
||||||
|
**both** `per_capita = TRUE` and `adjust_to_year` are set -- there is no
|
||||||
|
nominal per-capita column to deflate otherwise). Amounts are full US
|
||||||
|
dollars.
|
||||||
|
|
||||||
|
Carries a `provenance` attribute matching
|
||||||
|
`inst/schemas/provenance-v1.json`, whose `balance_caveats` block reports
|
||||||
|
`not_gaap`, `not_gaap_note`, `coverage_window` (measured year extents for
|
||||||
|
every balance subtype in the mounted corpus, not only the observed ones)
|
||||||
|
and `truncated` (the observed subtypes whose coverage falls short of the
|
||||||
|
requested years). `expenditure_concept`/`revenue_concept` are `NA` --
|
||||||
|
holdings are a stock, not a flow, so neither concept vocabulary applies.
|
||||||
|
}
|
||||||
|
\description{
|
||||||
|
Returns Census cash-and-security holdings (`category_type = "balance"`):
|
||||||
|
fund balances, retirement system holdings and insurance trust balances.
|
||||||
|
}
|
||||||
|
\section{Holdings are not GAAP fund balance}{
|
||||||
|
|
||||||
|
Census holdings are **gross** -- no liabilities are netted -- so a reserve
|
||||||
|
ratio built from them overstates what is actually available. They are not
|
||||||
|
comparable to a GAAP fund balance from an ACFR.
|
||||||
|
}
|
||||||
|
|
||||||
+15
-4
@@ -7,8 +7,8 @@
|
|||||||
cog_categories(type = NULL, pattern = NULL)
|
cog_categories(type = NULL, pattern = NULL)
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
\item{type}{Either `NULL` (default, return both spending and revenue
|
\item{type}{Either `NULL` (default, every row: expenditure, revenue and
|
||||||
rows), `"spending"`, or `"revenue"`.}
|
balance), `"spending"`, `"revenue"`, or `"balance"`.}
|
||||||
|
|
||||||
\item{pattern}{Optional regex matched case-insensitively against the
|
\item{pattern}{Optional regex matched case-insensitively against the
|
||||||
`category` column (e.g. `"Police"` or `"Tax"`).}
|
`category` column (e.g. `"Police"` or `"Tax"`).}
|
||||||
@@ -16,13 +16,24 @@ rows), `"spending"`, or `"revenue"`.}
|
|||||||
\value{
|
\value{
|
||||||
Tibble with columns `category`, `category_type`, `subtype`,
|
Tibble with columns `category`, `category_type`, `subtype`,
|
||||||
`n_codes`, `item_codes` (comma-separated, alphabetical). Sorted by
|
`n_codes`, `item_codes` (comma-separated, alphabetical). Sorted by
|
||||||
`category_type`, `category`, `subtype`.
|
`category_type`, `category`, `subtype`. Includes one row per flow for the
|
||||||
|
reserved pseudo-category `"All Categories"`, which carries `NA` for
|
||||||
|
`subtype`, `n_codes` and `item_codes` because it is a query mode rather
|
||||||
|
than a crosswalk entry — see [cog_spending()]'s `category` argument.
|
||||||
}
|
}
|
||||||
\description{
|
\description{
|
||||||
Returns the category taxonomy exposed by the corpus's
|
Returns the category taxonomy exposed by the corpus's
|
||||||
`summary_categories` view, grouped to one row per
|
`summary_categories` view, grouped to one row per
|
||||||
`(category, subtype)` pair. Use this to discover valid `category`
|
`(category, subtype)` pair. Use this to discover valid `category`
|
||||||
values for [cog_spending()] / [cog_revenue()] /
|
values for [cog_spending()] / [cog_revenue()] / [cog_balances()] /
|
||||||
[cog_geographic_rollup()] and to audit which Census item codes feed
|
[cog_geographic_rollup()] and to audit which Census item codes feed
|
||||||
each category.
|
each category.
|
||||||
}
|
}
|
||||||
|
\details{
|
||||||
|
`subtype` COALESCEs the crosswalk's three subtype columns, so it carries
|
||||||
|
`spend_subtype` on expenditure rows, `revenue_subtype` on revenue rows and
|
||||||
|
`balance_subtype` on balance rows. Note that [cog_balances()] itself takes
|
||||||
|
no `subtype` argument — for holdings, `category` is a strict coarsening of
|
||||||
|
`balance_subtype` — but the value is surfaced here because it is the
|
||||||
|
discovery surface downstream consumers build their vocabulary from.
|
||||||
|
}
|
||||||
|
|||||||
@@ -21,3 +21,33 @@ Prints the structured provenance attached to a tibble returned by any
|
|||||||
`cog_*` verb, or returns it as a list for downstream use (MCP tools,
|
`cog_*` verb, or returns it as a list for downstream use (MCP tools,
|
||||||
dashboards, JSON export).
|
dashboards, JSON export).
|
||||||
}
|
}
|
||||||
|
\section{Two kinds of series break}{
|
||||||
|
|
||||||
|
Catalogued breaks reach you without being asked for, in two disjoint
|
||||||
|
fields, because a caveat about one series and a caveat about the whole
|
||||||
|
corpus are different claims:
|
||||||
|
|
||||||
|
* **`series_break_refs`** — breaks matched against the item codes actually
|
||||||
|
present in this result. A break in one code you queried.
|
||||||
|
* **`corpus_break_refs`** — breaks catalogued with `fin_code = "ALL"`,
|
||||||
|
which are statements about the corpus rather than about any one code:
|
||||||
|
dollar precision across the 1976/1977 boundary (`SB085`), imputation
|
||||||
|
exclusion from FY2002 (`SB087`), the FY2012 dense-to-sparse
|
||||||
|
representation change (`SB194`), and the FY2017 government-identifier
|
||||||
|
change (`SB086`). These are selected on the break-year window alone.
|
||||||
|
|
||||||
|
`SB194` is the one most likely to matter: a query spanning FY2011 to FY2012
|
||||||
|
crosses the boundary where an absent cell stops meaning "Census published
|
||||||
|
$0" and starts meaning "not reported".
|
||||||
|
}
|
||||||
|
|
||||||
|
\section{Other provenance blocks}{
|
||||||
|
|
||||||
|
`transformations$units_conversion` records the `$1,000s`-to-dollars
|
||||||
|
multiply that every amount column has already had applied.
|
||||||
|
`transformations$per_capita` records the population denominator and its
|
||||||
|
year range. `coverage` and `coverage_mode` appear on multi-government
|
||||||
|
results (see [cog_geographic_rollup()]). `completion` appears when
|
||||||
|
`complete = TRUE`. `balance_caveats` appears on [cog_balances()] results.
|
||||||
|
}
|
||||||
|
|
||||||
|
|||||||
+21
-10
@@ -6,17 +6,22 @@
|
|||||||
\usage{
|
\usage{
|
||||||
cog_find_peers(
|
cog_find_peers(
|
||||||
target_govid,
|
target_govid,
|
||||||
|
year = NULL,
|
||||||
same_type = TRUE,
|
same_type = TRUE,
|
||||||
same_state = FALSE,
|
same_state = FALSE,
|
||||||
pop_range = c(0.7, 1.3),
|
pop_range = c(0.7, 1.3),
|
||||||
is_ratio = TRUE,
|
is_ratio = TRUE,
|
||||||
pop_year = NULL,
|
max_peers = 10L,
|
||||||
max_peers = 10L
|
coverage = c("all", "census", "consistent")
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
\item{target_govid}{Character scalar — `canonical_govid` of the target.}
|
\item{target_govid}{Character scalar — `canonical_govid` of the target.}
|
||||||
|
|
||||||
|
\item{year}{Integer scalar. Cohort vintage. When `NULL` (default), uses the
|
||||||
|
most recent year for which the target has an observed population in
|
||||||
|
`gov_population_yearly`.}
|
||||||
|
|
||||||
\item{same_type}{If `TRUE` (default) restrict peers to the target's
|
\item{same_type}{If `TRUE` (default) restrict peers to the target's
|
||||||
`govs_type`.}
|
`govs_type`.}
|
||||||
|
|
||||||
@@ -26,20 +31,26 @@ Default `FALSE`.}
|
|||||||
\item{pop_range}{Length-2 numeric vector giving lower/upper bounds.}
|
\item{pop_range}{Length-2 numeric vector giving lower/upper bounds.}
|
||||||
|
|
||||||
\item{is_ratio}{If `TRUE` (default) `pop_range` is multiplied by the
|
\item{is_ratio}{If `TRUE` (default) `pop_range` is multiplied by the
|
||||||
target's `population_acs` to produce absolute bounds. If `FALSE`,
|
target's population at `year` to produce absolute bounds. If `FALSE`,
|
||||||
`pop_range` is interpreted as absolute population counts.}
|
`pop_range` is interpreted as absolute population counts.}
|
||||||
|
|
||||||
\item{pop_year}{Reserved for future use (selecting ACS vintage). Currently
|
|
||||||
the corpus has a single snapshot so this argument has no effect.}
|
|
||||||
|
|
||||||
\item{max_peers}{Integer cap on the number of peers returned.}
|
\item{max_peers}{Integer cap on the number of peers returned.}
|
||||||
|
|
||||||
|
\item{coverage}{Survey-cycle handling; see [cog_peer_compare()]. Here it
|
||||||
|
governs the cohort VINTAGE when `year` is `NULL`: `"census"` snaps to the
|
||||||
|
most recent census year with an observed population, so a cohort is not
|
||||||
|
built from a sample year in which most of the candidate universe is
|
||||||
|
absent. `"consistent"` needs a year range, which cohort selection does not
|
||||||
|
have, so it selects like `"all"` and is carried on the result as
|
||||||
|
`attr(x, "coverage")` for [cog_peer_compare()].}
|
||||||
}
|
}
|
||||||
\value{
|
\value{
|
||||||
Tibble with columns `canonical_govid`, `gov_name`, `fips_state`,
|
Tibble with columns `canonical_govid`, `gov_name`, `fips_state`,
|
||||||
`population_acs`, `pop_ratio`, `rank`.
|
`population`, `pop_ratio`, `rank`. The cohort year is attached as
|
||||||
|
`attr(x, "cohort_year")`.
|
||||||
}
|
}
|
||||||
\description{
|
\description{
|
||||||
Selects peer governments from `canonical_fips_xwalk` by combinations of
|
Selects peer governments by combinations of government type, state, and
|
||||||
government type, state, and population range. Peers are ordered by
|
population range at a chosen `year`. Peers are ordered by `|log(pop_ratio)|`
|
||||||
`|log(pop_ratio)|` ascending (closest to the target's population first).
|
ascending (closest to the target's population first).
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -9,7 +9,9 @@ cog_geographic_rollup(
|
|||||||
category,
|
category,
|
||||||
years,
|
years,
|
||||||
per_capita = FALSE,
|
per_capita = FALSE,
|
||||||
adjust_to_year = NULL
|
adjust_to_year = NULL,
|
||||||
|
expenditure_concept = c("primary", "direct", "total"),
|
||||||
|
coverage = c("all", "census", "consistent")
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
@@ -18,21 +20,54 @@ cog_geographic_rollup(
|
|||||||
`canonical_govid` values. At least one layer required.}
|
`canonical_govid` values. At least one layer required.}
|
||||||
|
|
||||||
\item{category}{Single category name or character vector (passed through
|
\item{category}{Single category name or character vector (passed through
|
||||||
to [cog_spending()]).}
|
to [cog_spending()]), or the reserved `"All Categories"` for one summed
|
||||||
|
row per `(year, canonical_govid, subtype)` covering every category in the
|
||||||
|
concept's scope. `"All Categories"` is the efficient way to build a
|
||||||
|
geographic total: without it a caller must issue one rollup per category
|
||||||
|
and sum the results themselves.}
|
||||||
|
|
||||||
\item{years}{Integer vector of years.}
|
\item{years}{Integer vector of years.}
|
||||||
|
|
||||||
\item{per_capita}{If `TRUE`, per-capita uses each layer's own population
|
\item{per_capita}{If `TRUE`, per-capita uses each gov's own per-year
|
||||||
from `canonical_fips_xwalk.population_acs`.}
|
population from `gov_population_yearly`. Govs with missing population
|
||||||
|
are excluded from the result.}
|
||||||
|
|
||||||
\item{adjust_to_year}{Integer base year for CPI-U conversion, or `NULL`.}
|
\item{adjust_to_year}{Integer base year for CPI-U conversion, or `NULL`.}
|
||||||
|
|
||||||
|
\item{expenditure_concept}{`"primary"` (default), `"direct"`, or
|
||||||
|
`"total"` -- see [cog_spending()] for the three concepts. `"total"` is
|
||||||
|
refused here because combining Total across multiple layers of
|
||||||
|
government double-counts intergovernmental transfers (a state's payment
|
||||||
|
to a school district is the same dollar the district reports as its own
|
||||||
|
Direct spending); `"primary"` and `"direct"` combine safely.}
|
||||||
|
|
||||||
|
\item{coverage}{How to handle the Census of Governments survey cycle,
|
||||||
|
which is a **complete census only in years ending in 2 and 7** -- every
|
||||||
|
other year is a sample, and the sample varies enormously (on the bundled
|
||||||
|
fixture, Wisconsin's 608-city universe reports 597 governments in FY2012
|
||||||
|
and 112 in FY2019).
|
||||||
|
|
||||||
|
* `"all"` (default) -- every unit that reported that year. Unchanged
|
||||||
|
behaviour, so existing code keeps working.
|
||||||
|
* `"census"` -- census years only. Aborts if the requested range holds
|
||||||
|
none, rather than silently returning nothing.
|
||||||
|
* `"consistent"` -- only units reporting in *every* requested year, giving
|
||||||
|
a balanced panel.
|
||||||
|
|
||||||
|
Regardless of mode, `provenance$coverage` always carries per-year
|
||||||
|
`n_units_reporting`, `n_units_expected` and `is_census_year`, and
|
||||||
|
`provenance$coverage_mode` records the mode. `is_census_year` is a
|
||||||
|
statement about the **survey calendar**, never a claim of completeness:
|
||||||
|
FY1967 is a census year in which only 97 of Wisconsin's 608 cities
|
||||||
|
report. `n_units_reporting` is the number that tells the truth.}
|
||||||
}
|
}
|
||||||
\value{
|
\value{
|
||||||
Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`,
|
Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`,
|
||||||
`spend_subtype`, `category`, `amt_nominal`, optional `amt_real` /
|
`spend_subtype`, `category`, `amt_nominal`, optional `amt_real` /
|
||||||
`amt_per_capita_nominal` / `amt_per_capita_real`, `codes_included`,
|
`amt_per_capita_nominal` / `amt_per_capita_real`, optional `pop_source`,
|
||||||
`aggregate_fallback`, `scope_note`, `notes`. Carries a `provenance`
|
`codes_included`, `aggregate_fallback`, `scope_note`, `notes`. Carries a
|
||||||
attribute with `verb = "cog_geographic_rollup"` and `layers`.
|
`provenance` attribute with `verb = "cog_geographic_rollup"`, `layers`,
|
||||||
|
and `rollup$included_govids` / `rollup$excluded_govids`.
|
||||||
}
|
}
|
||||||
\description{
|
\description{
|
||||||
Wraps [cog_spending()], tags each row with its layer, and attaches a
|
Wraps [cog_spending()], tags each row with its layer, and attaches a
|
||||||
@@ -41,3 +76,31 @@ human-readable `scope_note` documenting geographic-scope caveats (e.g.
|
|||||||
"place portraits" that compare a city to the surrounding county and
|
"place portraits" that compare a city to the surrounding county and
|
||||||
containing state on one set of axes.
|
containing state on one set of axes.
|
||||||
}
|
}
|
||||||
|
\details{
|
||||||
|
When `per_capita = TRUE`, rows whose government has no observed
|
||||||
|
population in that year (`pop_source == "unavailable"`) are dropped from
|
||||||
|
the result. The dropped govids are recorded in
|
||||||
|
`provenance$rollup$excluded_govids`. This excludes special districts
|
||||||
|
(gov type 4) and school districts (gov type 5) from per-capita rollups
|
||||||
|
by design — see `vignette('population-denominators')`.
|
||||||
|
}
|
||||||
|
\section{Reading `coverage`}{
|
||||||
|
|
||||||
|
`provenance$coverage` reports `n_units_reporting` against
|
||||||
|
`n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
||||||
|
it counts governments with rows for the category you asked for, not
|
||||||
|
governments collected that year.** A government that was surveyed and
|
||||||
|
genuinely spends nothing in that category is indistinguishable here from one
|
||||||
|
that was never surveyed.
|
||||||
|
|
||||||
|
The ratio is therefore **not a response rate** and must not be used as one.
|
||||||
|
In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||||
|
`category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||||
|
contract policing to the county sheriff, not non-response.
|
||||||
|
|
||||||
|
The comparison that *is* valid is the same category across a census year
|
||||||
|
(ending in 2 or 7) and a sample year, where the real-zero component is
|
||||||
|
roughly constant and the difference reflects the survey cycle. `is_census_year`
|
||||||
|
marks which is which.
|
||||||
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -32,8 +32,11 @@ the cross-vintage canonical-government registry. Operates in two modes:
|
|||||||
}
|
}
|
||||||
\details{
|
\details{
|
||||||
* **Utility mode** (single `name`, the original behavior): returns all
|
* **Utility mode** (single `name`, the original behavior): returns all
|
||||||
rows whose `gov_name` matches the regex case-insensitively, sorted by
|
rows whose `gov_name` contains `name` as a **literal, case-insensitive
|
||||||
`population_acs` descending. Useful for exploratory lookups.
|
substring**, sorted by `population_acs` descending. Useful for
|
||||||
|
exploratory lookups. Regex metacharacters in `name` are escaped, so a
|
||||||
|
government is findable by its own complete name even when that name
|
||||||
|
contains parentheses or a period.
|
||||||
* **Basket mode** (`length(name) > 1`): resolves each input row to a
|
* **Basket mode** (`length(name) > 1`): resolves each input row to a
|
||||||
single canonical govid and returns a tibble in input order, suitable
|
single canonical govid and returns a tibble in input order, suitable
|
||||||
for piping straight into [cog_spending()] / [cog_revenue()] /
|
for piping straight into [cog_spending()] / [cog_revenue()] /
|
||||||
@@ -45,7 +48,8 @@ the cross-vintage canonical-government registry. Operates in two modes:
|
|||||||
1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`.
|
1. Filter `canonical_fips_xwalk` by `state` and (if non-NA) `type`.
|
||||||
2. **Exact pass:** case-insensitive equality against `gov_name`.
|
2. **Exact pass:** case-insensitive equality against `gov_name`.
|
||||||
Single hit -> resolved. Multiple -> step 4.
|
Single hit -> resolved. Multiple -> step 4.
|
||||||
3. **Substring fallback:** case-insensitive regex against `gov_name`.
|
3. **Substring fallback:** case-insensitive literal substring against
|
||||||
|
`gov_name` (metacharacters escaped).
|
||||||
Single hit -> resolved (`match_method = "substring"`). Zero hits ->
|
Single hit -> resolved (`match_method = "substring"`). Zero hits ->
|
||||||
`status = "no_match"`. Multiple hits -> step 4.
|
`status = "no_match"`. Multiple hits -> step 4.
|
||||||
4. **Disambiguation:** if matches share one `govs_type`, pick the
|
4. **Disambiguation:** if matches share one `govs_type`, pick the
|
||||||
@@ -58,7 +62,7 @@ inputs (`ambiguous` / `no_match`) appear only in the sidecar.
|
|||||||
}
|
}
|
||||||
\examples{
|
\examples{
|
||||||
\dontrun{
|
\dontrun{
|
||||||
# Utility mode — exploratory regex lookup
|
# Utility mode — exploratory substring lookup
|
||||||
cog_gov_search("broward", state = "FL")
|
cog_gov_search("broward", state = "FL")
|
||||||
|
|
||||||
# Basket mode — resolve a known cohort
|
# Basket mode — resolve a known cohort
|
||||||
|
|||||||
@@ -0,0 +1,18 @@
|
|||||||
|
% Generated by roxygen2: do not edit by hand
|
||||||
|
% Please edit documentation in R/manifest.R
|
||||||
|
\name{cog_manifest}
|
||||||
|
\alias{cog_manifest}
|
||||||
|
\title{Return the parsed corpus manifest for the active session.}
|
||||||
|
\usage{
|
||||||
|
cog_manifest()
|
||||||
|
}
|
||||||
|
\value{
|
||||||
|
Named list: `schema_version`, `built_at`, `pipeline_commit`,
|
||||||
|
`data_vintage`, `scope`, `years` (schema v5+), `schema`, `files`.
|
||||||
|
}
|
||||||
|
\description{
|
||||||
|
Opens a session (connecting to the configured corpus) if none is active,
|
||||||
|
then returns the manifest exactly as parsed from `manifest.json`. Useful
|
||||||
|
for consumers that need the published year range (`years` block, schema
|
||||||
|
v5+) or the partition list without issuing a data query.
|
||||||
|
}
|
||||||
+90
-5
@@ -10,7 +10,9 @@ cog_peer_compare(
|
|||||||
category,
|
category,
|
||||||
years,
|
years,
|
||||||
per_capita = TRUE,
|
per_capita = TRUE,
|
||||||
adjust_to_year = NULL
|
adjust_to_year = NULL,
|
||||||
|
expenditure_concept = c("primary", "direct", "total"),
|
||||||
|
coverage = c("all", "census", "consistent")
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
@@ -27,18 +29,101 @@ cog_peer_compare(
|
|||||||
population.}
|
population.}
|
||||||
|
|
||||||
\item{adjust_to_year}{Integer base year for CPI-U conversion or `NULL`.}
|
\item{adjust_to_year}{Integer base year for CPI-U conversion or `NULL`.}
|
||||||
|
|
||||||
|
\item{expenditure_concept}{`"primary"` (default), `"direct"`, or
|
||||||
|
`"total"` -- see [cog_spending()] for the three concepts. `"total"` is
|
||||||
|
refused here because combining Total across peer sets counts
|
||||||
|
intergovernmental transfers twice; `"primary"` and `"direct"` combine
|
||||||
|
safely.}
|
||||||
|
|
||||||
|
\item{coverage}{How to handle the Census of Governments survey cycle,
|
||||||
|
which is a **complete census only in years ending in 2 and 7** -- every
|
||||||
|
other year is a sample, and the sample varies enormously (on the bundled
|
||||||
|
fixture, Wisconsin's 608-city universe reports 597 governments in FY2012
|
||||||
|
and 112 in FY2019).
|
||||||
|
|
||||||
|
* `"all"` (default) -- every unit that reported that year. Unchanged
|
||||||
|
behaviour, so existing code keeps working.
|
||||||
|
* `"census"` -- census years only. Aborts if the requested range holds
|
||||||
|
none, rather than silently returning nothing.
|
||||||
|
* `"consistent"` -- only units reporting in *every* requested year, giving
|
||||||
|
a balanced panel.
|
||||||
|
|
||||||
|
Regardless of mode, `provenance$coverage` always carries per-year
|
||||||
|
`n_units_reporting`, `n_units_expected` and `is_census_year`, and
|
||||||
|
`provenance$coverage_mode` records the mode. `is_census_year` is a
|
||||||
|
statement about the **survey calendar**, never a claim of completeness:
|
||||||
|
FY1967 is a census year in which only 97 of Wisconsin's 608 cities
|
||||||
|
report. `n_units_reporting` is the number that tells the truth.
|
||||||
|
|
||||||
|
The comparison target is exempt from `"consistent"` balancing -- it is the
|
||||||
|
subject of the comparison, not a member of the cohort -- and the
|
||||||
|
`summary_*` quantiles are computed AFTER the filter, so they describe the
|
||||||
|
cohort actually returned. `n_units_reporting` counts peers only, against
|
||||||
|
the cohort size: "3 of your 15 peers reported in FY2019".}
|
||||||
}
|
}
|
||||||
\value{
|
\value{
|
||||||
Tibble matching [cog_spending()]'s columns, plus a `role`
|
Tibble matching [cog_spending()]'s columns, plus a `role`
|
||||||
column taking values `"target"`, `"peer"`, `"summary_p25"`,
|
column taking values `"target"`, `"peer"`, `"summary_p25"`,
|
||||||
`"summary_p50"`, or `"summary_p75"`, and `target_rank` (target's rank
|
`"summary_p50"`, or `"summary_p75"`, `target_rank` (target's rank
|
||||||
among target+peers at `max(years)`, NA for other rows). Provenance
|
among target+peers at `max(years)`, NA for other rows), and
|
||||||
attribute reports `verb = "cog_peer_compare"` and `peer_count`.
|
`cohort_year` (the year used to build the peer cohort, read from
|
||||||
|
`attr(peers, "cohort_year")`; `NA` when `peers` was a bare character
|
||||||
|
vector). Provenance reports `verb = "cog_peer_compare"`, `peer_count`,
|
||||||
|
`cohort_year`, and `cohort_govids`.
|
||||||
|
|
||||||
|
**The `summary_*` rows are per-category quantiles: they are not additive.**
|
||||||
|
Each one is computed **within each `(year, spend_subtype,
|
||||||
|
category)` cell** across the peer set, so a `summary_p50` row is *the
|
||||||
|
median peer's value in that one category*, not *the value of the median
|
||||||
|
peer's total*. The median peer for Police and the median peer for Fire
|
||||||
|
are usually different governments, so summing `summary_*` rows across
|
||||||
|
categories does not give any peer's total and misstates the band it
|
||||||
|
appears to describe — measured at −32.7% to +251.0% across 24 years on
|
||||||
|
one cohort, with a sign flip at FY2012.
|
||||||
|
|
||||||
|
Facet by `role` **and** `category` (the documented use, and what the
|
||||||
|
rows are built for). For a genuine "median peer's total spending" line,
|
||||||
|
sum each peer's own categories first and take the quantile of those
|
||||||
|
per-government totals:
|
||||||
|
|
||||||
|
```r
|
||||||
|
library(dplyr)
|
||||||
|
cmp |>
|
||||||
|
filter(role %in% c("target", "peer")) |>
|
||||||
|
group_by(year, role, canonical_govid) |>
|
||||||
|
summarise(total = sum(amt_per_capita_real, na.rm = TRUE), .groups = "drop") |>
|
||||||
|
filter(role == "peer") |>
|
||||||
|
group_by(year) |>
|
||||||
|
summarise(p50 = quantile(total, 0.5, na.rm = TRUE))
|
||||||
|
```
|
||||||
}
|
}
|
||||||
\description{
|
\description{
|
||||||
Pulls spending for the target plus a peer set (either a
|
Pulls spending for the target plus a peer set (either a
|
||||||
[cog_find_peers()] result or a character vector of `canonical_govid`) and
|
[cog_find_peers()] result or a character vector of `canonical_govid`) and
|
||||||
appends peer-distribution summary rows (`summary_p25`, `summary_p50`,
|
appends peer-distribution summary rows (`summary_p25`, `summary_p50`,
|
||||||
`summary_p75`) so the result can be faceted by `role` in a single ggplot
|
`summary_p75`) so the result can be faceted by `role` in a single ggplot
|
||||||
call.
|
call. Those summary rows are quantiles **within each category**, not
|
||||||
|
quantiles of each peer's total — see the `@return` section before summing
|
||||||
|
them.
|
||||||
}
|
}
|
||||||
|
\section{Reading `coverage`}{
|
||||||
|
|
||||||
|
`provenance$coverage` reports `n_units_reporting` against
|
||||||
|
`n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
||||||
|
it counts cohort members with rows for the category you asked for, not
|
||||||
|
cohort members collected that year.** A government that was surveyed and
|
||||||
|
genuinely spends nothing in that category is indistinguishable here from one
|
||||||
|
that was never surveyed.
|
||||||
|
|
||||||
|
The ratio is therefore **not a response rate** and must not be used as one.
|
||||||
|
In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||||
|
`category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||||
|
contract policing to the county sheriff, not non-response.
|
||||||
|
|
||||||
|
The comparison that *is* valid is the same category across a census year
|
||||||
|
(ending in 2 or 7) and a sample year, where the real-zero component is
|
||||||
|
roughly constant and the difference reflects the survey cycle. `is_census_year`
|
||||||
|
marks which is which.
|
||||||
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,22 @@
|
|||||||
|
% Generated by roxygen2: do not edit by hand
|
||||||
|
% Please edit documentation in R/recipes.R
|
||||||
|
\name{cog_recipes}
|
||||||
|
\alias{cog_recipes}
|
||||||
|
\title{List available harmonization recipes}
|
||||||
|
\usage{
|
||||||
|
cog_recipes(pattern = NULL)
|
||||||
|
}
|
||||||
|
\arguments{
|
||||||
|
\item{pattern}{Optional regex matched case-insensitively against
|
||||||
|
`recipe_id` or `label`.}
|
||||||
|
}
|
||||||
|
\value{
|
||||||
|
Tibble with columns `recipe_id`, `label`, `n_components`,
|
||||||
|
`year_min`, `year_max` (the min/max component year coverage), sorted by
|
||||||
|
`recipe_id`.
|
||||||
|
}
|
||||||
|
\description{
|
||||||
|
Recipes are multi-code cross-vintage series (see [cog_spending()]'s
|
||||||
|
`recipe` argument) catalogued in the corpus's `harmonization_recipes`
|
||||||
|
table. Use this to discover valid `recipe` ids.
|
||||||
|
}
|
||||||
+136
-7
@@ -5,33 +5,162 @@
|
|||||||
\title{Summarized revenue by category}
|
\title{Summarized revenue by category}
|
||||||
\usage{
|
\usage{
|
||||||
cog_revenue(
|
cog_revenue(
|
||||||
govid,
|
govid = NULL,
|
||||||
years,
|
years,
|
||||||
category = NULL,
|
category = NULL,
|
||||||
per_capita = FALSE,
|
per_capita = FALSE,
|
||||||
adjust_to_year = NULL
|
adjust_to_year = NULL,
|
||||||
|
basis = c("harmonized", "raw"),
|
||||||
|
recipe = NULL,
|
||||||
|
revenue_concept = c("general", "total"),
|
||||||
|
complete = FALSE,
|
||||||
|
limit = NULL,
|
||||||
|
offset = NULL,
|
||||||
|
state = NULL,
|
||||||
|
type = NULL
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
\item{govid}{Character vector of `canonical_govid` values.}
|
\item{govid}{Character vector of `canonical_govid` values, or `NULL` to name
|
||||||
|
the cohort by `state`/`type` instead. One of `govid`, `state`, or `type`
|
||||||
|
is required.}
|
||||||
|
|
||||||
\item{years}{Integer vector of years.}
|
\item{years}{Integer vector of years.}
|
||||||
|
|
||||||
\item{category}{Character vector of category names (from
|
\item{category}{Character vector of category names (from
|
||||||
`summary_categories.category`), or `NULL` for all categories.}
|
`summary_categories.category`), or `NULL` for all categories broken out
|
||||||
|
one row each. The reserved value `"All Categories"` instead returns a
|
||||||
|
single summed row per `(year, canonical_govid, subtype)`, covering every
|
||||||
|
category inside the requested concept's subtype scope. It cannot be
|
||||||
|
combined with other category names, and it is not the same thing as
|
||||||
|
`revenue_concept = "total"`: the concept chooses which subtypes are in
|
||||||
|
scope, `"All Categories"` chooses whether rows inside that scope are
|
||||||
|
broken out or summed. Because the result keeps one row per
|
||||||
|
`revenue_subtype`, filtering the returned frame to
|
||||||
|
`revenue_subtype == "own_source"` gives an own-source revenue total.}
|
||||||
|
|
||||||
\item{per_capita}{If `TRUE`, adds `amt_per_capita_nominal` (and
|
\item{per_capita}{If `TRUE`, adds `amt_per_capita_nominal` (and
|
||||||
`amt_per_capita_real` when `adjust_to_year` is set) using
|
`amt_per_capita_real` when `adjust_to_year` is set) using the per-year
|
||||||
`population_acs` from the canonical xwalk.}
|
Census F-33 population from `gov_population_yearly`. Result also gains
|
||||||
|
a `pop_source` column with values `"census_f33"` or `"unavailable"`
|
||||||
|
(the latter for gov types 4/5 and any row whose population is missing
|
||||||
|
in that year).}
|
||||||
|
|
||||||
\item{adjust_to_year}{Integer base year for CPI-U real-dollar conversion,
|
\item{adjust_to_year}{Integer base year for CPI-U real-dollar conversion,
|
||||||
or `NULL` for nominal only.}
|
or `NULL` for nominal only.}
|
||||||
|
|
||||||
|
\item{basis}{`"harmonized"` (default) sums item codes through the
|
||||||
|
cross-vintage harmonization mapping (folding series-break-affected
|
||||||
|
codes onto a comparable target and excluding aggregate / discontinued
|
||||||
|
rows -- see the `harmonization` block in `cog_explain()`); `"raw"`
|
||||||
|
reproduces the pre-Phase-R2 behavior (published item codes, no
|
||||||
|
folding). On a corpus with `schema_version < 5` (no harmonization
|
||||||
|
tables), `basis` silently resolves to `"raw"` when left at its default
|
||||||
|
and the resolution is recorded in the provenance; explicitly passing
|
||||||
|
`basis = "harmonized"` on such a corpus aborts. Ignored when `recipe`
|
||||||
|
is set (see below).}
|
||||||
|
|
||||||
|
\item{recipe}{Optional harmonization recipe id (see [cog_recipes()]) for
|
||||||
|
multi-code cross-vintage series that a 1:1 harmonized_code mapping
|
||||||
|
can't express (e.g. a wide-era aggregate that only splits into leaf
|
||||||
|
codes in the modern era). Mutually exclusive with `category`. The
|
||||||
|
result's subtype column reads `"recipe"` and `category` reads the
|
||||||
|
recipe's label. Requires `schema_version >= 5`. A recipe query bypasses
|
||||||
|
`basis` entirely (it joins `long` directly rather than going through
|
||||||
|
the `*_annotated`/`*_annotated_harmonized` views), so the `basis`
|
||||||
|
argument is ignored and the result's provenance reports
|
||||||
|
`basis = "recipe"` with an inert `harmonization` block (`applied =
|
||||||
|
FALSE`, pointing at the `recipe` block instead) rather than a
|
||||||
|
possibly-misleading `"harmonized"`/`"raw"` value.}
|
||||||
|
|
||||||
|
\item{revenue_concept}{Which of Census's two published revenue concepts to
|
||||||
|
return. Concepts are defined as sets of the crosswalk's `revenue_subtype`
|
||||||
|
values -- never as item-code first letters, which cannot classify
|
||||||
|
correctly (prefix `Y` spans revenue, expenditure and balance codes, and
|
||||||
|
prefix `X` does the same):
|
||||||
|
|
||||||
|
* `"general"` (default) -- Census General Revenue: `own_source` +
|
||||||
|
`federal` + `state` + `local_aid`. The manual defines this concept by
|
||||||
|
subtraction (section 4.3: *"General revenue comprises all revenue
|
||||||
|
except that classified as liquor store, utility, or insurance trust
|
||||||
|
revenue"*), so utility (`A91`-`A94`), liquor store (`A90`) and
|
||||||
|
insurance trust revenue are all excluded.
|
||||||
|
* `"total"` -- Census Total Revenue: every revenue subtype, i.e.
|
||||||
|
`general` plus utility, liquor store, and insurance trust revenue
|
||||||
|
(`Y01`/`Y02`/`Y04`/`Y11`/`Y12`/`Y51`/`Y52` and the employee-retirement
|
||||||
|
`X01`/`X02`/`X05`/`X08`).
|
||||||
|
|
||||||
|
The two are related by Census's own identity, `Total Revenue = General +
|
||||||
|
Utility + Liquor Store + Insurance Trust`.
|
||||||
|
|
||||||
|
Note that the employee-retirement (`X`) codes stop at FY2016, when those
|
||||||
|
systems moved out of the annual finance file into the separate Annual
|
||||||
|
Survey of Public Pensions, so a `"total"` series steps down at the
|
||||||
|
FY2016/FY2017 seam for reasons that are about collection scope rather
|
||||||
|
than revenue (series breaks `SB197`-`SB202`).}
|
||||||
|
|
||||||
|
\item{complete}{If `TRUE`, fill the requested grid so that a cell the
|
||||||
|
corpus does not carry still appears, labelled with **why** it is
|
||||||
|
missing, and add a `value_source` column to every row:
|
||||||
|
|
||||||
|
* `"reported"` — the corpus carries this cell.
|
||||||
|
* `"census_zero"` — dense-source year (`<= FY2011`), cell absent:
|
||||||
|
Census published `$0`. `amt_nominal` is `0`.
|
||||||
|
* `"not_reported"` — sparse-source year (`>= FY2012`), cell absent: the
|
||||||
|
government did not report, and the value is unknown. `amt_nominal` is
|
||||||
|
`NA`, **not** `0` — writing a zero there would invent data.
|
||||||
|
|
||||||
|
The grid comes from the corpus's `code_set` table, scoped to each
|
||||||
|
government's own type, so a county is never filled with cells only a
|
||||||
|
state can report. Reported rows are passed through untouched.
|
||||||
|
|
||||||
|
Defaults to `FALSE` (the historical behaviour: absent cells simply do
|
||||||
|
not appear). Needs a corpus published from 2026-07-29 onward, which is
|
||||||
|
when `representation`/`code_set` began shipping; aborts with class
|
||||||
|
`uscogdata_representation_unavailable` otherwise. Not available with
|
||||||
|
`recipe` or with `expenditure_concept = "total"` (class
|
||||||
|
`uscogdata_complete_unsupported`) — neither draws its cells from
|
||||||
|
`code_set`.}
|
||||||
|
|
||||||
|
\item{limit}{Maximum number of result rows to return, pushed into the SQL
|
||||||
|
query itself (`LIMIT`/`OFFSET`) rather than applied after the full
|
||||||
|
result is materialized. `NULL` (the default) returns every matching row,
|
||||||
|
exactly as before this parameter existed. Mutually exclusive with
|
||||||
|
`recipe` and with `complete = TRUE` -- see `offset` and `total_rows`.}
|
||||||
|
|
||||||
|
\item{offset}{Rows to skip before `limit` starts counting (0-based).
|
||||||
|
Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set.}
|
||||||
|
|
||||||
|
\item{state, type}{Name the cohort by predicate instead of by id: `state` is
|
||||||
|
a 2-letter USPS abbreviation (or a FIPS code) and `type` is one of
|
||||||
|
`"state"`, `"county"`, `"city"`, `"township"` (or the integer `0:3`) --
|
||||||
|
the same vocabulary, and the same internal coercion, as
|
||||||
|
[cog_gov_search()]. Both default to `NULL`.
|
||||||
|
|
||||||
|
The cohort is then expressed as a subquery against `canonical_fips_xwalk`
|
||||||
|
inside each statement rather than round-tripped through R as a literal id
|
||||||
|
list. For a fleet-scale cohort that is the difference between a
|
||||||
|
301,591-character `IN` list re-parsed in 5--8 statements per call and a
|
||||||
|
constant-size predicate: measured at **94 ms versus 449 ms** for the same
|
||||||
|
FY2022 aggregate over the 20,106-government `type = "city"` cohort, within
|
||||||
|
7% of the no-filter floor.
|
||||||
|
|
||||||
|
Supplying `govid` **and** `state`/`type` INTERSECTS them -- the
|
||||||
|
governments in `govid` that also match the predicate -- rather than one
|
||||||
|
silently taking precedence. Naming no cohort at all (`govid`, `state` and
|
||||||
|
`type` all `NULL`) aborts with class `uscogdata_no_cohort`.
|
||||||
|
|
||||||
|
When the cohort is named by predicate, `provenance$scope$govids_found`
|
||||||
|
and `govids_missing` are empty -- there is no id list to report against --
|
||||||
|
and `provenance$scope$cohort` carries `state`, `type` and
|
||||||
|
`n_governments` instead. A `govid`-named cohort reports exactly as before.}
|
||||||
}
|
}
|
||||||
\value{
|
\value{
|
||||||
Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||||
`revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
`revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
||||||
optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
||||||
`codes_included`, `aggregate_fallback`, `notes`.
|
optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`,
|
||||||
|
and `value_source` when `complete = TRUE`.
|
||||||
}
|
}
|
||||||
\description{
|
\description{
|
||||||
Mirror of [cog_spending()] for revenue categories. One row per
|
Mirror of [cog_spending()] for revenue categories. One row per
|
||||||
|
|||||||
+163
-8
@@ -5,34 +5,189 @@
|
|||||||
\title{Summarized spending by category}
|
\title{Summarized spending by category}
|
||||||
\usage{
|
\usage{
|
||||||
cog_spending(
|
cog_spending(
|
||||||
govid,
|
govid = NULL,
|
||||||
years,
|
years,
|
||||||
category = NULL,
|
category = NULL,
|
||||||
per_capita = FALSE,
|
per_capita = FALSE,
|
||||||
adjust_to_year = NULL
|
adjust_to_year = NULL,
|
||||||
|
basis = c("harmonized", "raw"),
|
||||||
|
recipe = NULL,
|
||||||
|
expenditure_concept = c("primary", "direct", "total"),
|
||||||
|
complete = FALSE,
|
||||||
|
limit = NULL,
|
||||||
|
offset = NULL,
|
||||||
|
state = NULL,
|
||||||
|
type = NULL
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
\item{govid}{Character vector of `canonical_govid` values.}
|
\item{govid}{Character vector of `canonical_govid` values, or `NULL` to name
|
||||||
|
the cohort by `state`/`type` instead. One of `govid`, `state`, or `type`
|
||||||
|
is required.}
|
||||||
|
|
||||||
\item{years}{Integer vector of years.}
|
\item{years}{Integer vector of years.}
|
||||||
|
|
||||||
\item{category}{Character vector of category names (from
|
\item{category}{Character vector of category names (from
|
||||||
`summary_categories.category`), or `NULL` for all categories.}
|
`summary_categories.category`), or `NULL` for all categories broken out
|
||||||
|
one row each. The reserved value `"All Categories"` instead returns a
|
||||||
|
single summed row per `(year, canonical_govid, subtype)`, covering every
|
||||||
|
category inside the requested concept's subtype scope. It cannot be
|
||||||
|
combined with other category names, and it is not the same thing as
|
||||||
|
`expenditure_concept = "total"`: the concept chooses which subtypes are in
|
||||||
|
scope, `"All Categories"` chooses whether rows inside that scope are
|
||||||
|
broken out or summed. Because the result keeps one row per
|
||||||
|
`spend_subtype`, filtering the returned frame to
|
||||||
|
`spend_subtype == "operations"` gives an operating-expenditure total.}
|
||||||
|
|
||||||
\item{per_capita}{If `TRUE`, adds `amt_per_capita_nominal` (and
|
\item{per_capita}{If `TRUE`, adds `amt_per_capita_nominal` (and
|
||||||
`amt_per_capita_real` when `adjust_to_year` is set) using
|
`amt_per_capita_real` when `adjust_to_year` is set) using the per-year
|
||||||
`population_acs` from the canonical xwalk.}
|
Census F-33 population from `gov_population_yearly`. Result also gains
|
||||||
|
a `pop_source` column with values `"census_f33"` or `"unavailable"`
|
||||||
|
(the latter for gov types 4/5 and any row whose population is missing
|
||||||
|
in that year).}
|
||||||
|
|
||||||
\item{adjust_to_year}{Integer base year for CPI-U real-dollar conversion,
|
\item{adjust_to_year}{Integer base year for CPI-U real-dollar conversion,
|
||||||
or `NULL` for nominal only.}
|
or `NULL` for nominal only.}
|
||||||
|
|
||||||
|
\item{basis}{`"harmonized"` (default) sums item codes through the
|
||||||
|
cross-vintage harmonization mapping (folding series-break-affected
|
||||||
|
codes onto a comparable target and excluding aggregate / discontinued
|
||||||
|
rows -- see the `harmonization` block in `cog_explain()`); `"raw"`
|
||||||
|
reproduces the pre-Phase-R2 behavior (published item codes, no
|
||||||
|
folding). On a corpus with `schema_version < 5` (no harmonization
|
||||||
|
tables), `basis` silently resolves to `"raw"` when left at its default
|
||||||
|
and the resolution is recorded in the provenance; explicitly passing
|
||||||
|
`basis = "harmonized"` on such a corpus aborts. Ignored when `recipe`
|
||||||
|
is set (see below).}
|
||||||
|
|
||||||
|
\item{recipe}{Optional harmonization recipe id (see [cog_recipes()]) for
|
||||||
|
multi-code cross-vintage series that a 1:1 harmonized_code mapping
|
||||||
|
can't express (e.g. a wide-era aggregate that only splits into leaf
|
||||||
|
codes in the modern era). Mutually exclusive with `category`. The
|
||||||
|
result's subtype column reads `"recipe"` and `category` reads the
|
||||||
|
recipe's label. Requires `schema_version >= 5`. A recipe query bypasses
|
||||||
|
`basis` entirely (it joins `long` directly rather than going through
|
||||||
|
the `*_annotated`/`*_annotated_harmonized` views), so the `basis`
|
||||||
|
argument is ignored and the result's provenance reports
|
||||||
|
`basis = "recipe"` with an inert `harmonization` block (`applied =
|
||||||
|
FALSE`, pointing at the `recipe` block instead) rather than a
|
||||||
|
possibly-misleading `"harmonized"`/`"raw"` value.}
|
||||||
|
|
||||||
|
\item{expenditure_concept}{Which spending concept to return. Concepts are
|
||||||
|
defined as sets of the crosswalk's `spend_subtype` values -- never as
|
||||||
|
item-code first letters, which cannot classify correctly (prefix `Y`
|
||||||
|
alone spans revenue, expenditure, and balance codes):
|
||||||
|
|
||||||
|
* `"primary"` (default) -- the government's own service provision:
|
||||||
|
`operations` + `capital` + `assistance` subtypes.
|
||||||
|
* `"direct"` -- Census's published Direct Expenditure: `primary` plus
|
||||||
|
`interest` (interest on debt) and `insurance_benefits` (insurance
|
||||||
|
trust benefit payments, e.g. pensions -- Census manual section
|
||||||
|
5.2.2.1 includes payments to retirees in Direct).
|
||||||
|
* `"total"` -- `direct` plus the intergovernmental leg: payments to
|
||||||
|
local governments (`M` codes), to the state government (`L` codes,
|
||||||
|
excluding the `L--` family-total rollup), and state payments to
|
||||||
|
school systems (`Q11`/`Q12`/`Q18`), so results gain rows with
|
||||||
|
`spend_subtype == "intergovernmental"`. Requires the active corpus's
|
||||||
|
`summary_categories` to carry M/L rows (added by cog_pipeline PR
|
||||||
|
#59); aborts with class `uscogdata_ig_categories_unsupported` on an
|
||||||
|
older corpus rather than silently under-reporting. Mutually
|
||||||
|
exclusive with `recipe` (a recipe already defines its own component
|
||||||
|
codes).
|
||||||
|
|
||||||
|
**Do not sum `"total"` results across levels of government** (e.g.
|
||||||
|
state + county + city): a state's `M12` payment to a school district is
|
||||||
|
the same dollar the district reports as its own direct `E12`, so
|
||||||
|
summing both double-counts it. This matters in particular with
|
||||||
|
[cog_geographic_rollup()], which sums across exactly that kind of
|
||||||
|
multi-layer government set.
|
||||||
|
|
||||||
|
In the legacy wide era (<= FY2011), some functions are published ONLY
|
||||||
|
as an aggregate-flagged family total (e.g. Corrections' `E04`/`E05`
|
||||||
|
split), which the Direct leg excludes by construction but the IG leg
|
||||||
|
deliberately keeps (see `inst/sql/24-ig_long.sql`). For a `"total"`
|
||||||
|
query, any (year, category) where this leaves intergovernmental rows
|
||||||
|
with NO Direct counterpart is flagged: the affected rows' `notes`
|
||||||
|
name the harmonization recipe that recovers the missing Direct
|
||||||
|
component (when one exists), and
|
||||||
|
`provenance$expenditure_concept_direct_suppressed` is `TRUE` -- the
|
||||||
|
figure in those rows is the intergovernmental leg alone, not Direct +
|
||||||
|
IG. When `category = "All Categories"` is combined with
|
||||||
|
`expenditure_concept = "total"`, this detection cannot run (it keys on
|
||||||
|
per-category rows, which all-categories mode collapses to one literal
|
||||||
|
value), so `expenditure_concept_direct_suppressed` is `NA` rather than a
|
||||||
|
possibly-false `FALSE`; query an explicit `category` to get a real
|
||||||
|
answer.}
|
||||||
|
|
||||||
|
\item{complete}{If `TRUE`, fill the requested grid so that a cell the
|
||||||
|
corpus does not carry still appears, labelled with **why** it is
|
||||||
|
missing, and add a `value_source` column to every row:
|
||||||
|
|
||||||
|
* `"reported"` — the corpus carries this cell.
|
||||||
|
* `"census_zero"` — dense-source year (`<= FY2011`), cell absent:
|
||||||
|
Census published `$0`. `amt_nominal` is `0`.
|
||||||
|
* `"not_reported"` — sparse-source year (`>= FY2012`), cell absent: the
|
||||||
|
government did not report, and the value is unknown. `amt_nominal` is
|
||||||
|
`NA`, **not** `0` — writing a zero there would invent data.
|
||||||
|
|
||||||
|
The grid comes from the corpus's `code_set` table, scoped to each
|
||||||
|
government's own type, so a county is never filled with cells only a
|
||||||
|
state can report. Reported rows are passed through untouched.
|
||||||
|
|
||||||
|
Defaults to `FALSE` (the historical behaviour: absent cells simply do
|
||||||
|
not appear). Needs a corpus published from 2026-07-29 onward, which is
|
||||||
|
when `representation`/`code_set` began shipping; aborts with class
|
||||||
|
`uscogdata_representation_unavailable` otherwise. Not available with
|
||||||
|
`recipe` or with `expenditure_concept = "total"` (class
|
||||||
|
`uscogdata_complete_unsupported`) — neither draws its cells from
|
||||||
|
`code_set`.}
|
||||||
|
|
||||||
|
\item{limit}{Maximum number of result rows to return, pushed into the SQL
|
||||||
|
query itself (`LIMIT`/`OFFSET`) rather than applied after the full
|
||||||
|
result is materialized. `NULL` (the default) returns every matching row,
|
||||||
|
exactly as before this parameter existed. Mutually exclusive with
|
||||||
|
`recipe` and with `complete = TRUE` -- see `offset` and `total_rows`.}
|
||||||
|
|
||||||
|
\item{offset}{Rows to skip before `limit` starts counting (0-based).
|
||||||
|
Ignored if `limit` is `NULL`; defaults to `0L` when `limit` is set.}
|
||||||
|
|
||||||
|
\item{state, type}{Name the cohort by predicate instead of by id: `state` is
|
||||||
|
a 2-letter USPS abbreviation (or a FIPS code) and `type` is one of
|
||||||
|
`"state"`, `"county"`, `"city"`, `"township"` (or the integer `0:3`) --
|
||||||
|
the same vocabulary, and the same internal coercion, as
|
||||||
|
[cog_gov_search()]. Both default to `NULL`.
|
||||||
|
|
||||||
|
The cohort is then expressed as a subquery against `canonical_fips_xwalk`
|
||||||
|
inside each statement rather than round-tripped through R as a literal id
|
||||||
|
list. For a fleet-scale cohort that is the difference between a
|
||||||
|
301,591-character `IN` list re-parsed in 5--8 statements per call and a
|
||||||
|
constant-size predicate: measured at **94 ms versus 449 ms** for the same
|
||||||
|
FY2022 aggregate over the 20,106-government `type = "city"` cohort, within
|
||||||
|
7% of the no-filter floor.
|
||||||
|
|
||||||
|
Supplying `govid` **and** `state`/`type` INTERSECTS them -- the
|
||||||
|
governments in `govid` that also match the predicate -- rather than one
|
||||||
|
silently taking precedence. Naming no cohort at all (`govid`, `state` and
|
||||||
|
`type` all `NULL`) aborts with class `uscogdata_no_cohort`.
|
||||||
|
|
||||||
|
When the cohort is named by predicate, `provenance$scope$govids_found`
|
||||||
|
and `govids_missing` are empty -- there is no id list to report against --
|
||||||
|
and `provenance$scope$cohort` carries `state`, `type` and
|
||||||
|
`n_governments` instead. A `govid`-named cohort reports exactly as before.}
|
||||||
}
|
}
|
||||||
\value{
|
\value{
|
||||||
Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||||
`spend_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
`spend_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
||||||
optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
||||||
`codes_included`, `aggregate_fallback`, `notes`. Carries a `provenance`
|
optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`,
|
||||||
attribute matching `inst/schemas/provenance-v1.json`.
|
and `value_source` when `complete = TRUE`.
|
||||||
|
Carries a `provenance` attribute matching `inst/schemas/provenance-v1.json`,
|
||||||
|
whose `completion` block reports `applied`, `rows_filled`, and the
|
||||||
|
per-year `absence_means` rule that was applied. When `limit` is set,
|
||||||
|
also carries a `total_rows` attribute: the full unpaginated row count,
|
||||||
|
computed by the same query (`COUNT(*) OVER()`) rather than a second
|
||||||
|
round trip -- so a caller walking pages never has to ask "how many are
|
||||||
|
there" separately.
|
||||||
}
|
}
|
||||||
\description{
|
\description{
|
||||||
One row per `(year, canonical_govid, spend_subtype, category)`. Amounts are
|
One row per `(year, canonical_govid, spend_subtype, category)`. Amounts are
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,285 @@
|
|||||||
|
# Per-year population denominators in uscogdata
|
||||||
|
|
||||||
|
**Date:** 2026-04-29
|
||||||
|
**Status:** Design — pending implementation
|
||||||
|
**Scope:** uscogdata 0.1 (pre-release; no version bump)
|
||||||
|
**Related:** cog_pipeline (data dictionary updates)
|
||||||
|
|
||||||
|
## Problem
|
||||||
|
|
||||||
|
`uscogdata::cog_spending(per_capita = TRUE)` and `cog_revenue(per_capita = TRUE)`
|
||||||
|
currently divide every year's nominal amount by a single static population value
|
||||||
|
— `canonical_fips_xwalk.population_acs`, the ACS 2018-2022 5-year estimate.
|
||||||
|
|
||||||
|
For a 24-year corpus (2000–2023) this introduces a systematic bias proportional
|
||||||
|
to each government's population change over that span. Fast-growing places have
|
||||||
|
their early-year per-capita numbers understated; shrinking places have theirs
|
||||||
|
overstated. The bias commonly exceeds 20% and can exceed 50% for cities like
|
||||||
|
Detroit. Provenance currently advertises this denominator explicitly, so the
|
||||||
|
error is visible to careful users — but the default behavior produces wrong
|
||||||
|
numbers.
|
||||||
|
|
||||||
|
`cog_geographic_rollup()` has the same bug. `cog_find_peers()` /
|
||||||
|
`cog_peer_compare()` use the same static value to define peer cohorts, which
|
||||||
|
is defensible for matching but is no longer necessary now that per-year
|
||||||
|
population is available.
|
||||||
|
|
||||||
|
## Background — population sources
|
||||||
|
|
||||||
|
| Source | What it is | Where it lives |
|
||||||
|
|---|---|---|
|
||||||
|
| **Census F-33 `population`** | Population value Census uses on each COG row to compute its own per-capita tables. Almost always a Population Estimates Program (PEP) estimate; sometimes lagged a year for fiscal-year alignment, recorded in `popyear` | `long.population`, `long.popyear` (per row) |
|
||||||
|
| **PEP** (raw) | Census Bureau's official annual intercensal estimates. Distinct from F-33 because F-33 sometimes uses a lagged vintage | Not in corpus; available via tidycensus |
|
||||||
|
| **ACS 5-year** | American Community Survey 5-year rolling average. Different methodology, includes margin of error, only available 2005-2009 onward | `canonical_fips_xwalk.population_acs` (one fixed vintage) |
|
||||||
|
| **Decennial** | Actual count, every 10 years | Not in corpus |
|
||||||
|
|
||||||
|
F-33 `population` is the right default: it's what Census itself uses, so per-
|
||||||
|
capita results published by uscogdata reconcile with Census's own published
|
||||||
|
tables.
|
||||||
|
|
||||||
|
## Approach
|
||||||
|
|
||||||
|
Use the per-row `population` already present in `long`, joined on
|
||||||
|
`(canonical_govid, year)`. No new external data dependency. Coverage:
|
||||||
|
|
||||||
|
- **Types 0–3** (state, county, city, township): observed every year by design
|
||||||
|
- **Types 4–5** (special districts, schools): always NA — masked in
|
||||||
|
`cog_pipeline/R/read_modern.R` because the F-33 schema does not carry a
|
||||||
|
population value for these gov types
|
||||||
|
|
||||||
|
Type-4 and type-5 govs return `NA` per-capita with a `pop_source = "unavailable"`
|
||||||
|
flag and a note. No silent substitution.
|
||||||
|
|
||||||
|
The architecture leaves the door open for future denominators (PEP, ACS,
|
||||||
|
decennial) by surfacing `pop_source` as a first-class result column. Adding a
|
||||||
|
new source later is a join change, not an API change.
|
||||||
|
|
||||||
|
## Detailed design
|
||||||
|
|
||||||
|
### New view: `gov_population_yearly`
|
||||||
|
|
||||||
|
```sql
|
||||||
|
-- inst/sql/32-gov_population_yearly.sql
|
||||||
|
CREATE OR REPLACE VIEW gov_population_yearly AS
|
||||||
|
SELECT DISTINCT
|
||||||
|
year,
|
||||||
|
canonical_govid,
|
||||||
|
population,
|
||||||
|
popyear
|
||||||
|
FROM long
|
||||||
|
WHERE population IS NOT NULL;
|
||||||
|
```
|
||||||
|
|
||||||
|
`SELECT DISTINCT` collapses the metadata column duplicated across each gov-year's
|
||||||
|
item rows. A test asserts `(year, canonical_govid)` is unique to catch any
|
||||||
|
future source-data divergence.
|
||||||
|
|
||||||
|
### `cog_spending()` and `cog_revenue()`
|
||||||
|
|
||||||
|
`.attach_per_capita()` (in `R/spending.R`) is rewritten to:
|
||||||
|
|
||||||
|
1. Query `gov_population_yearly` for the requested govids and years.
|
||||||
|
2. `LEFT JOIN` on `(canonical_govid, year)` so missing rows produce NA.
|
||||||
|
3. Compute `amt_per_capita_nominal = amt_nominal / population`. NA when
|
||||||
|
population is NA.
|
||||||
|
4. Drop `population` from the returned tibble (keep `pop_source` instead).
|
||||||
|
|
||||||
|
Result tibble gains one new column when `per_capita = TRUE`:
|
||||||
|
|
||||||
|
- `pop_source`: `"census_f33"` when a denominator was found, `"unavailable"`
|
||||||
|
when NA.
|
||||||
|
|
||||||
|
`notes` is extended: when `pop_source == "unavailable"`, append
|
||||||
|
`"No population denominator available for this gov type"`. The `notes` column
|
||||||
|
is updated to concatenate multiple notes with `"; "` (it currently holds at
|
||||||
|
most one).
|
||||||
|
|
||||||
|
`amt_per_capita_real` is NA whenever `amt_per_capita_nominal` is NA.
|
||||||
|
|
||||||
|
### `cog_geographic_rollup()`
|
||||||
|
|
||||||
|
The current implementation does **not** sum amounts within a layer — it returns
|
||||||
|
one row per `(year, canonical_govid, subtype, category)` tagged with its
|
||||||
|
layer, intended for side-by-side "place portrait" comparisons (a city, the
|
||||||
|
county containing it, the state containing both). That semantics is preserved.
|
||||||
|
|
||||||
|
The only behavior change in this work is per-row exclusion when `per_capita = TRUE`:
|
||||||
|
|
||||||
|
1. After `cog_spending()` returns with the per-row per-year denominator from
|
||||||
|
Task 3, drop rows where `pop_source == "unavailable"` so the result never
|
||||||
|
contains NA per-capita rows.
|
||||||
|
2. Record the dropped `canonical_govid`s in `provenance$rollup$excluded_govids`
|
||||||
|
and the kept ones in `provenance$rollup$included_govids`.
|
||||||
|
|
||||||
|
Documentation states explicitly: *Per-capita rollups include only governments
|
||||||
|
observed in both the finance and population panels for the given year. Special
|
||||||
|
districts and school districts (gov types 4 and 5) are therefore excluded from
|
||||||
|
per-capita rollups by design.*
|
||||||
|
|
||||||
|
Provenance gains:
|
||||||
|
|
||||||
|
- `rollup.included_govids` — `canonical_govid`s present in the result
|
||||||
|
- `rollup.excluded_govids` — `canonical_govid`s dropped for missing pop
|
||||||
|
|
||||||
|
### `cog_find_peers()`
|
||||||
|
|
||||||
|
Signature: `cog_find_peers(target_govid, year = NULL, pop_range = c(0.5, 2), ...)`
|
||||||
|
|
||||||
|
- `year` is a single integer. When `NULL`, defaults to the most recent year
|
||||||
|
present in `gov_population_yearly` for the target.
|
||||||
|
- Looks up target's `population` at `year`. Errors if NA, with a message
|
||||||
|
listing nearby years where target *is* observed.
|
||||||
|
- Filters candidates by `gov_population_yearly.population` at the same `year`,
|
||||||
|
within `pop_range[1] * target_pop` and `pop_range[2] * target_pop`.
|
||||||
|
- Orders by `|log(pop_ratio)|` ascending.
|
||||||
|
|
||||||
|
Returned columns: `canonical_govid`, `gov_name`, `govs_type`, `fips_state`,
|
||||||
|
`population`, `pop_ratio`, `rank`. The column previously named `population_acs`
|
||||||
|
is renamed to `population`.
|
||||||
|
|
||||||
|
The cohort year is attached as a tibble attribute: `attr(x, "cohort_year")`.
|
||||||
|
|
||||||
|
### `cog_peer_compare()`
|
||||||
|
|
||||||
|
Existing signature unchanged:
|
||||||
|
`cog_peer_compare(target_govid, peers, category, years, per_capita = TRUE, adjust_to_year = NULL)`.
|
||||||
|
The caller supplies `peers` (either a `cog_find_peers()` result tibble or a
|
||||||
|
character vector of `canonical_govid`). The cohort year is implicit in
|
||||||
|
whichever year the caller used to call `cog_find_peers()`.
|
||||||
|
|
||||||
|
Behavior changes:
|
||||||
|
|
||||||
|
- When `peers` is a tibble carrying `attr(peers, "cohort_year")`,
|
||||||
|
`cog_peer_compare()` reads it and stamps every result row with a constant
|
||||||
|
`cohort_year` column.
|
||||||
|
- When `peers` is a bare character vector, `cohort_year` in the result is `NA`.
|
||||||
|
- Provenance gets `cohort_year` (scalar or NA) and the cohort govids list.
|
||||||
|
|
||||||
|
Users who want time-varying cohorts call `cog_find_peers()` per year and
|
||||||
|
stitch the `cog_peer_compare()` results themselves — documented in the
|
||||||
|
vignette with a worked example.
|
||||||
|
|
||||||
|
### Provenance updates
|
||||||
|
|
||||||
|
`provenance$transformations$per_capita` becomes:
|
||||||
|
|
||||||
|
```r
|
||||||
|
list(
|
||||||
|
applied = TRUE,
|
||||||
|
denominator_source = "Census F-33 population (per-year, from long.population)",
|
||||||
|
popyear_range = c(<min>, <max>),
|
||||||
|
pop_source_counts = list(census_f33 = N1, unavailable = N2)
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
For peer compare results, additional provenance:
|
||||||
|
|
||||||
|
```r
|
||||||
|
list(
|
||||||
|
cohort_year = <int>,
|
||||||
|
cohort_govids = <character>,
|
||||||
|
pop_range = c(<lo>, <hi>)
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
For rollup results, additional provenance:
|
||||||
|
|
||||||
|
```r
|
||||||
|
list(
|
||||||
|
rollup = list(
|
||||||
|
included_govids = <character>,
|
||||||
|
excluded_govids = <character>
|
||||||
|
)
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
`R/explain.R` is updated to render the new fields.
|
||||||
|
|
||||||
|
### Documentation
|
||||||
|
|
||||||
|
**New vignette** `vignettes/population-denominators.Rmd`:
|
||||||
|
|
||||||
|
1. The four population sources explained
|
||||||
|
2. Why F-33 is the default — and how it reconciles with Census's own per-capita
|
||||||
|
tables
|
||||||
|
3. The `popyear` quirk: Census sometimes uses a lagged estimate for fiscal-year
|
||||||
|
alignment. Recorded in provenance, not in the result.
|
||||||
|
4. Worked example showing the bias from the old static-ACS approach versus
|
||||||
|
per-year F-33 (e.g., Detroit 2003 vs. 2023)
|
||||||
|
5. Worked example of a rolling-cohort peer comparison built by looping
|
||||||
|
`cog_peer_compare()` per year
|
||||||
|
6. Future direction: `pop_source` is structured so PEP, ACS time-series, or
|
||||||
|
decennial denominators can be added later without API changes
|
||||||
|
|
||||||
|
**`cog_pipeline/docs/data_dictionary.md`** entry for `long.population` and
|
||||||
|
`long.popyear`: definition, source (F-33 fixed-width files, byte ranges),
|
||||||
|
type-4/5 masking rule, relationship to PEP.
|
||||||
|
|
||||||
|
### Tests
|
||||||
|
|
||||||
|
- `gov_population_yearly` returns one row per `(year, canonical_govid)` (uniqueness)
|
||||||
|
- `cog_spending(per_capita = TRUE)` returns different denominators for
|
||||||
|
different years for a known gov in the fixture (use any gov whose population
|
||||||
|
changes between 2019 and 2020)
|
||||||
|
- Type-4 and type-5 govids in the fixture return `pop_source = "unavailable"`
|
||||||
|
and `NA` per-capita with the expected note
|
||||||
|
- `cog_geographic_rollup(per_capita = TRUE)` excludes missing-pop govs and
|
||||||
|
records them in provenance
|
||||||
|
- `cog_find_peers()` defaults `year` to the most recent year for a target
|
||||||
|
with known population history
|
||||||
|
- `cog_find_peers()` errors with a helpful message when target has no observed
|
||||||
|
population in the requested year
|
||||||
|
- `cog_peer_compare()` defaults `cohort_year` and produces a result with a
|
||||||
|
constant `cohort_year` column
|
||||||
|
- Provenance carries `denominator_source`, `popyear_range`, and
|
||||||
|
`pop_source_counts`
|
||||||
|
- Regression test against a fixed govid+year showing the new per-capita value
|
||||||
|
differs from the old (static-ACS) by exactly the ratio of `population_acs`
|
||||||
|
to `long.population` for that gov-year
|
||||||
|
|
||||||
|
### Migration
|
||||||
|
|
||||||
|
Pre-release; no version bump. `NEWS.md` Unreleased entry:
|
||||||
|
|
||||||
|
> **Per-capita denominators now use per-year Census F-33 population.**
|
||||||
|
> Previously, `cog_spending()` and `cog_revenue()` divided all years' amounts
|
||||||
|
> by a single ACS 2018-2022 population, producing biased per-capita values
|
||||||
|
> for time-series. They now divide by the F-33 `population` recorded for each
|
||||||
|
> gov-year. Type-4 (special districts) and type-5 (school districts) govs
|
||||||
|
> return `NA` per-capita with `pop_source = "unavailable"`.
|
||||||
|
>
|
||||||
|
> **Peer matching now uses per-year population.** `cog_find_peers()` gains a
|
||||||
|
> `year` argument (defaults to most recent observed year). `cog_peer_compare()`
|
||||||
|
> gains `cohort_year`. Cohorts are still fixed for a single peer-compare call;
|
||||||
|
> users wanting moving cohorts loop themselves.
|
||||||
|
>
|
||||||
|
> **Rollups exclude govs with missing population.** `cog_geographic_rollup()`
|
||||||
|
> per-capita totals include only govs where both the finance variable and
|
||||||
|
> population are observed in that year; excluded govids are recorded in
|
||||||
|
> provenance.
|
||||||
|
>
|
||||||
|
> Returned column `population_acs` from `cog_find_peers()` is renamed to
|
||||||
|
> `population` and reflects the cohort-year vintage.
|
||||||
|
|
||||||
|
### File impact
|
||||||
|
|
||||||
|
| File | Change |
|
||||||
|
|---|---|
|
||||||
|
| `inst/sql/32-gov_population_yearly.sql` | New |
|
||||||
|
| `R/spending.R` (`.attach_per_capita`, `.notes_column`) | Per-year join, `pop_source`, multi-note concat |
|
||||||
|
| `R/peers.R` (`cog_find_peers`, `cog_peer_compare`) | `year` / `cohort_year` args, query new view, column rename |
|
||||||
|
| `R/rollup.R` | Skip-with-record for missing-pop govs |
|
||||||
|
| `R/provenance.R` | New denominator/cohort/rollup fields |
|
||||||
|
| `R/explain.R` | Render new fields |
|
||||||
|
| `vignettes/population-denominators.Rmd` | New |
|
||||||
|
| `tests/testthat/` | Per-year denominator, type-4/5, rollup exclusion, peer cohort, provenance |
|
||||||
|
| `cog_pipeline/docs/data_dictionary.md` | Document `long.population`, `long.popyear`, masking |
|
||||||
|
| `NEWS.md` | Unreleased entry |
|
||||||
|
|
||||||
|
## Out of scope
|
||||||
|
|
||||||
|
- PEP/ACS/decennial denominators — architected for, not implemented
|
||||||
|
- `per_pupil` denominator using `long.enrollment` for type-5 — deferred
|
||||||
|
- Covering-county fallback for type-4 — deliberately not done
|
||||||
|
- Backfilling population for type-4/5 from any external source
|
||||||
|
- Changes to `cog_explorer` callers — separate follow-up, after this lands
|
||||||
@@ -0,0 +1,281 @@
|
|||||||
|
# `cog_balances()` — a reader surface for cash and security holdings
|
||||||
|
|
||||||
|
**Issue:** `uscogdata#25` requirement 2 · **Downstream:** `cog-api#26`
|
||||||
|
**Date:** 2026-08-03 · **Status:** design, awaiting approval
|
||||||
|
|
||||||
|
Requirement 1 of `uscogdata#25` (no `balance` row may reach a money verb) shipped
|
||||||
|
with `#11`/`#12` and is asserted at both view and verb level. This spec covers
|
||||||
|
requirement 2 only: a way to query holdings.
|
||||||
|
|
||||||
|
## Decision: a verb, not an argument
|
||||||
|
|
||||||
|
`cog_balances()`, parallel to `cog_spending()` / `cog_revenue()`.
|
||||||
|
|
||||||
|
Holdings are a **stock** — a balance at a point in time — while the money verbs
|
||||||
|
return **flows** over a fiscal year. The flow verbs' whole argument vocabulary
|
||||||
|
is meaningless for a stock: `expenditure_concept` / `revenue_concept` describe
|
||||||
|
which flows Census aggregates into a published total, and `complete=` fills a
|
||||||
|
grid of fiscal-year cells. Overloading a money verb would put a stock behind
|
||||||
|
arguments that all assume a flow.
|
||||||
|
|
||||||
|
## The 14 codes
|
||||||
|
|
||||||
|
Measured against the published corpus 2026-08-03, not transcribed from the
|
||||||
|
issue. `year_min`/`year_max` are observed row extents.
|
||||||
|
|
||||||
|
| `balance_subtype` | `category` | codes | observed years |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `general` | Fund Balances | `W01`, `W31`, `W61` | 2012–2021 |
|
||||||
|
| `employee_retirement` | Retirement System Holdings | `X21`, `X42`, `X44` | 1967–2016 |
|
||||||
|
| | | `X47` | 1988–2016 |
|
||||||
|
| | | `X30`, `Z77`, `Z78` | 2012–2016 |
|
||||||
|
| `unemployment_trust` | Insurance Trust Balances | `Y07`, `Y08` | 1967–2023 |
|
||||||
|
| `workers_comp_trust` | Insurance Trust Balances | `Y21` | 2012–2023 |
|
||||||
|
| `other_insurance_trust` | Insurance Trust Balances | `Y61` | 2012–2023 |
|
||||||
|
|
||||||
|
## Architecture
|
||||||
|
|
||||||
|
### Two new views
|
||||||
|
|
||||||
|
Mirroring the `revenue_long` / `revenue_annotated` pair exactly:
|
||||||
|
|
||||||
|
- `inst/sql/26-balance_long.sql` — `category_type = 'balance' AND NOT is_aggregate`
|
||||||
|
- `inst/sql/46-balance_annotated.sql` — joins `canonical_fips_xwalk` and
|
||||||
|
`summary_categories`, exposing `category`, `category_type`, `balance_subtype`
|
||||||
|
|
||||||
|
`.register_views()` globs `inst/sql/*.sql` in sorted order, so both register
|
||||||
|
with no new registration code.
|
||||||
|
|
||||||
|
### A third gate list in `R/views.R`
|
||||||
|
|
||||||
|
`CREATE VIEW` resolves its source schema eagerly, so a missing **column** fails
|
||||||
|
at registration time, not at query time. `46-balance_annotated.sql` selects
|
||||||
|
`c.balance_subtype`, which exists only on corpora built after pipeline `#76`/`#77`.
|
||||||
|
That arrived without a `schema_version` bump, so neither existing gate applies:
|
||||||
|
`.harmonization_view_files` keys on `schema_version`, `.representation_view_files`
|
||||||
|
on the presence of a *file*. The discriminator here is a **column on an existing
|
||||||
|
table**.
|
||||||
|
|
||||||
|
```r
|
||||||
|
.balance_view_files <- c("26-balance_long.sql", "46-balance_annotated.sql")
|
||||||
|
```
|
||||||
|
|
||||||
|
gated by probing `summary_categories` for `balance_subtype`, with
|
||||||
|
`cog_balances()` erroring cleanly via `.require_balance_support()` on an older
|
||||||
|
corpus — mirroring how `.require_schema_v5()` gates the harmonized views.
|
||||||
|
|
||||||
|
### `R/balances.R` — a dedicated path, not `.verb_spendrev()`
|
||||||
|
|
||||||
|
`.verb_spendrev()` is 825 lines whose concept scoping, intergovernmental leg and
|
||||||
|
`complete=` grid are all flow-specific, and four verbs depend on it. Threading a
|
||||||
|
third mode through it adds branching to shared code for no reuse benefit.
|
||||||
|
|
||||||
|
Reused unchanged: `.build_provenance()`, `.build_series_break_refs()`,
|
||||||
|
`.build_corpus_break_refs()`, the population join, `.inflate()`, and
|
||||||
|
`.coerce_govid_input()`.
|
||||||
|
|
||||||
|
Following the package's real two-layer convention: **view definitions** live in
|
||||||
|
`inst/sql/`; **query construction** is inline `sprintf()` in R, as in
|
||||||
|
`.verb_spendrev()`. (`CLAUDE.md` currently states "never inline SQL strings in R
|
||||||
|
files", which the verb layer has never obeyed. Corrected in a separate commit —
|
||||||
|
see Out of scope.)
|
||||||
|
|
||||||
|
## Signature
|
||||||
|
|
||||||
|
```r
|
||||||
|
cog_balances(govid, years,
|
||||||
|
category = NULL, # Fund Balances | Insurance Trust Balances |
|
||||||
|
# Retirement System Holdings
|
||||||
|
per_capita = FALSE,
|
||||||
|
adjust_to_year = NULL,
|
||||||
|
basis = c("harmonized", "raw"),
|
||||||
|
recipe = NULL)
|
||||||
|
```
|
||||||
|
|
||||||
|
Returns a `tbl_df` with a `provenance` attribute, like every other verb.
|
||||||
|
|
||||||
|
**Absent by design:** `expenditure_concept`, `revenue_concept`, `complete`,
|
||||||
|
and `subtype` — see below.
|
||||||
|
|
||||||
|
**`per_capita` is offered.** Holdings per resident is a real measure (pension
|
||||||
|
assets per capita, fund balance per resident). The roxygen `@param` states
|
||||||
|
plainly that this is a *stock per resident* and is **not** comparable to
|
||||||
|
`cog_spending()`'s per-capita figures.
|
||||||
|
|
||||||
|
**`basis` is currently a no-op** — `harmonization_map` has zero balance-code
|
||||||
|
rows, so harmonized and raw are identical for holdings. Kept for uniformity
|
||||||
|
with the money verbs (the API would otherwise special-case), and
|
||||||
|
`provenance$basis_note` says so outright rather than letting it look meaningful.
|
||||||
|
|
||||||
|
**`recipe` ships in v1 and works.** The two holdings recipes bridge the wide era
|
||||||
|
to the modern one:
|
||||||
|
|
||||||
|
```
|
||||||
|
cash_securities_z77_wide = X40 (1967-2011) + Z77 (2012-2023)
|
||||||
|
cash_securities_z78_wide = X41 (1967-2011) + Z78 (2012-2023)
|
||||||
|
```
|
||||||
|
|
||||||
|
`X40`/`X41` carry ~42,700 rows that are **100% `is_aggregate = TRUE`**, so they
|
||||||
|
are invisible to `balance_long`, which filters `NOT is_aggregate` like every
|
||||||
|
other basis view. That is by design, not a defect:
|
||||||
|
`cog_pipeline/docs/phase_r_harmonization_review.md` § 0.2 records that the wide
|
||||||
|
era exposes these split families *only* as aggregates, and that the recipe join
|
||||||
|
must therefore **not** filter `is_aggregate` — safe by construction, because
|
||||||
|
wide rows (≤2011) are aggregate-only, modern rows (2012+) are leaf-only, and
|
||||||
|
every component is year-scoped, so no double-count is possible. § 1 records the
|
||||||
|
matching decision that the planned `X40→Z77` harmonization *map* rows were
|
||||||
|
dropped and the continuity ships as recipes instead, which is why
|
||||||
|
`harmonization_map` has no balance-code rows.
|
||||||
|
|
||||||
|
The reader already implements this (`R/recipes.R`, `R/spending.R`), and it is
|
||||||
|
verified rather than assumed: `corrections_combined` for FY2007 — a recipe whose
|
||||||
|
wide leg `E05` is likewise aggregate-only — returns $906,743,000 against the
|
||||||
|
live corpus. So a recipe query reaches rows the verb's own view cannot, exactly
|
||||||
|
as intended.
|
||||||
|
|
||||||
|
### No `subtype` argument: `category` is a strict coarsening
|
||||||
|
|
||||||
|
`balance` is the only `category_type` in which `category` and the subtype column
|
||||||
|
are **not** orthogonal. Measured against the published crosswalk:
|
||||||
|
|
||||||
|
| `category_type` | subtypes spanning more than one category |
|
||||||
|
|---|---|
|
||||||
|
| expenditure | 5 of 6 (`operations`, `capital`, `interest`, `assistance`, `intergovernmental`) |
|
||||||
|
| revenue | 1 of 7 (`own_source`) |
|
||||||
|
| **balance** | **0 of 5** |
|
||||||
|
|
||||||
|
For expenditure the two axes are a genuine cross-tab — *function* (Police, Fire)
|
||||||
|
× *economic character* (operations, capital) — so both earn their place. For
|
||||||
|
balance the relation is a strict tree:
|
||||||
|
|
||||||
|
```
|
||||||
|
Fund Balances = {general} W01 W31 W61
|
||||||
|
Retirement System Holdings = {employee_retirement} X21 X30 X42 X44 X47 Z77 Z78
|
||||||
|
Insurance Trust Balances = {unemployment_trust,
|
||||||
|
workers_comp_trust,
|
||||||
|
other_insurance_trust} Y07 Y08 Y21 Y61
|
||||||
|
```
|
||||||
|
|
||||||
|
Exposing both would therefore admit no useful combination. Of the 15 possible
|
||||||
|
pairs, 3 are redundant (the subtype already implies its category) and **12 are
|
||||||
|
guaranteed empty for every government in every year** — and an impossible query
|
||||||
|
would fail by returning an empty tibble, which reads as "this government holds
|
||||||
|
none" rather than "you asked a contradiction."
|
||||||
|
|
||||||
|
Dropping `subtype` also keeps the verb aligned with the rest of the package: no
|
||||||
|
uscogdata verb exposes a subtype argument. `subtype_col` is internal plumbing in
|
||||||
|
`.verb_spendrev()`, and the API layers its own `subtype` row filter on top
|
||||||
|
(`api/R/handlers_governments.R`). `cog-api#26` can do exactly that for
|
||||||
|
`/balances`.
|
||||||
|
|
||||||
|
`#25`'s hard requirement is still met — `category = "Fund Balances"` *is* the
|
||||||
|
`general` family, precisely `W01`/`W31`/`W61`, in one filter. The only loss is
|
||||||
|
isolating one of the three insurance funds in a single argument;
|
||||||
|
`balance_subtype` remains a returned column, so that is one `dplyr::filter()`
|
||||||
|
away.
|
||||||
|
|
||||||
|
## Caveat surfacing
|
||||||
|
|
||||||
|
`provenance$balance_caveats`, always present, plus one `cli_inform()` per
|
||||||
|
session per caveat class when a query actually touches an affected family or
|
||||||
|
year. Structured so `cog-api#26` can forward the fields verbatim.
|
||||||
|
|
||||||
|
Verified against `series_breaks.csv`, not assumed:
|
||||||
|
|
||||||
|
| # | Caveat | Covered by existing machinery? |
|
||||||
|
|---|---|---|
|
||||||
|
| 1 | Gross holdings, **not GAAP fund balance**; no liabilities netted | No — a constant, new field `not_gaap = TRUE` |
|
||||||
|
| 2 | `W` is FY2012–2021 only | No — new `coverage_window`, **computed** from the corpus |
|
||||||
|
| 3 | `X`/`Z` holdings end FY2016 | **Not yet.** No `series_breaks` row exists at 2016/2017 for `Z77`/`Z78`/`X30`. Reader surfaces it via `coverage_window`; flows through `series_break_refs` once the upstream entry lands (see Out of scope) |
|
||||||
|
| 4 | `X40`/`X41` book → market at FY2002 | **Yes**, via `SB195`/`SB196` on `fin_code` `X40`/`X41`, under **two** conditions: a `recipe` query (the only path that observes those codes) **and** a year span that crosses FY2002. Asserted in the tests rather than assumed |
|
||||||
|
|
||||||
|
On caveat 4's second condition: `.build_series_break_refs()` matches
|
||||||
|
`break_year BETWEEN min(years) AND max(years)`, so a request spanning only
|
||||||
|
2011–2012 does **not** surface `SB195`. That is correct, not a gap — such a
|
||||||
|
series sits entirely after the change, on one consistent basis, and flagging a
|
||||||
|
break it never crosses would be noise. The same rule is applied deliberately in
|
||||||
|
`.build_corpus_break_refs()`. An earlier draft of this row omitted the span
|
||||||
|
condition and overclaimed.
|
||||||
|
|
||||||
|
`coverage_window` is derived per observed subtype family from the corpus, never
|
||||||
|
hardcoded, so it stays correct as the corpus grows.
|
||||||
|
|
||||||
|
`series_break_refs` and `corpus_break_refs` are otherwise populated by the
|
||||||
|
existing code-driven builders and need no change.
|
||||||
|
|
||||||
|
## Testing
|
||||||
|
|
||||||
|
New `tests/testthat/test-balances.R`. The bundled fixture covers all four
|
||||||
|
fixture years — `W` in 2012/2019/2020, the `X`/`Z` family in 2011/2012, `Y`
|
||||||
|
throughout — so every test below runs offline.
|
||||||
|
|
||||||
|
- **Inverse guard.** No flow code ever appears in `cog_balances()`, complementing
|
||||||
|
the already-asserted forward guard. Absence is verified against the raw corpus
|
||||||
|
via `read_parquet` on `data/long`, never through the verb that creates it.
|
||||||
|
- **FY2016 seam.** The `X`/`Z` family is present in 2012 and absent in 2019;
|
||||||
|
`coverage_window` reports the termination and the console message fires once.
|
||||||
|
- **Caveats.** `not_gaap` is always `TRUE`; `coverage_window` matches the
|
||||||
|
measured table above; the FY2002 valuation caveat fires only when the year
|
||||||
|
range crosses 2002 *and* touches `employee_retirement`.
|
||||||
|
- **`per_capita`.** `amt_per_capita_nominal == amt_nominal / population`.
|
||||||
|
- **`category = "Fund Balances"` is the `general` family.** Returns exactly
|
||||||
|
`W01`/`W31`/`W61` and nothing else — `#25`'s one-filter requirement, asserted
|
||||||
|
rather than assumed.
|
||||||
|
- **The hierarchy holds.** Every `balance_subtype` in the crosswalk maps to
|
||||||
|
exactly one `category`. Asserted against the crosswalk so that an upstream
|
||||||
|
change breaking the tree — which would silently make `category` lossy —
|
||||||
|
fails here rather than in a user's analysis.
|
||||||
|
- **`recipe` bridges the wide era.** `cash_securities_z77_wide` returns the
|
||||||
|
`X40` leg for a pre-2012 year, proving the aggregate-only wide rows are
|
||||||
|
reached — the property `phase_r_harmonization_review.md` § 0.2 depends on. A
|
||||||
|
regression here would silently truncate a 45-year series to five.
|
||||||
|
- **`SB195`/`SB196` reach the user on that path.** A `recipe` query spanning
|
||||||
|
FY2002 carries both in `provenance$series_break_refs`, so the book → market
|
||||||
|
basis change is disclosed wherever `X40`/`X41` are actually observed.
|
||||||
|
- **Gating.** `.require_balance_support()` errors cleanly on a corpus whose
|
||||||
|
`summary_categories` lacks `balance_subtype`.
|
||||||
|
|
||||||
|
## Out of scope, tracked separately
|
||||||
|
|
||||||
|
1. **Pipeline issue (new), non-blocking.** Catalogue the FY2016 termination of
|
||||||
|
the seven holdings codes in `series_breaks.csv`. There is currently **no**
|
||||||
|
entry at 2016/2017 for `Z77`/`Z78`/`X30`, although
|
||||||
|
`docs/phase_r_harmonization_review.md` § 2 identified the gap and recommended
|
||||||
|
exactly this — *"candidate new `series_breaks.csv` entries (recommend
|
||||||
|
`with_caution` documentation rows, no map action)"*. The follow-through never
|
||||||
|
happened. `SB197`–`SB202` set the precedent, giving the analogous X-flow
|
||||||
|
codes `coverage_restricted` + `with_caution` at 2017; `with_caution` is also
|
||||||
|
what keeps this out of the `joinable = "no"` identity-change rule, which
|
||||||
|
would otherwise oblige a harmonization-map row.
|
||||||
|
|
||||||
|
Verify the break corpus-wide and census-to-census before writing the rows.
|
||||||
|
`cog_balances()` does not wait on this — caveat 3 is covered reader-side by
|
||||||
|
`coverage_window` meanwhile, and the entry simply adds a second, catalogued
|
||||||
|
signpost when it lands.
|
||||||
|
|
||||||
|
**Superseded:** an earlier draft of this spec proposed adding
|
||||||
|
`summary_categories` rows for `X40`/`X41` and treated `recipe=` as blocked.
|
||||||
|
Both were wrong. `X40`/`X41` are deliberately aggregate-only per
|
||||||
|
`phase_r_harmonization_review.md` § 0.2, the dropped harmonization-map rows
|
||||||
|
are the documented § 1 decision, and the recipe path reaches them by design.
|
||||||
|
2. **`cog-api#26`.** Adds `/balances` in all three required places — handler,
|
||||||
|
`param_contract`, and the `plumber.R` route signature. Lands after this.
|
||||||
|
|
||||||
|
**Two contract facts the API must carry forward**, both settled during
|
||||||
|
implementation and easy to get wrong from the outside:
|
||||||
|
|
||||||
|
- `provenance$balance_caveats$coverage_window` is **corpus-scoped, not
|
||||||
|
result-scoped**. It reports the observed year extent of *every* balance
|
||||||
|
subtype in the corpus, not only the subtypes a given query returned — so a
|
||||||
|
`category = "Fund Balances"` query still returns all five windows. That is
|
||||||
|
deliberate: the windows describe what the corpus holds, which is what a
|
||||||
|
consumer needs in order to know what it did *not* ask for. The sibling
|
||||||
|
field `truncated` is the result-scoped one. Documented in
|
||||||
|
`inst/schemas/provenance-v1.json` and mutation-guarded against silent
|
||||||
|
inversion.
|
||||||
|
- `balance_caveats` appears **only** on `cog_balances()` results. It is
|
||||||
|
absent from `cog_spending()`/`cog_revenue()` provenance, and the schema
|
||||||
|
says so — an API layer that assumes it is universal will read `NULL`.
|
||||||
|
3. **`uscogdata/CLAUDE.md` refresh.** Separate commit. It is stale: it claims 7
|
||||||
|
SQL views (there are 21), 181 tests (716), a two-year fixture (four years),
|
||||||
|
and a "never inline SQL" rule the verb layer does not follow.
|
||||||
@@ -0,0 +1,296 @@
|
|||||||
|
# `uscogdata` 0.3.0 — public release
|
||||||
|
|
||||||
|
**Date:** 2026-08-08 · **Status:** design, awaiting approval
|
||||||
|
**Scope:** release-readiness, README, NEWS. Distribution mechanics recorded here as
|
||||||
|
decided, sequenced after the package is clean.
|
||||||
|
|
||||||
|
`uscogdata` is feature-complete and the corpus it reads has been public on
|
||||||
|
HuggingFace since 2026-08-07 (294 downloads as of this writing). The API built on
|
||||||
|
it is live. What does not exist is a public *package*: the repo is private, there
|
||||||
|
is no install path, and — measured, not assumed — **a stranger who installed it
|
||||||
|
today could not read the corpus at all.**
|
||||||
|
|
||||||
|
This spec covers making that untrue.
|
||||||
|
|
||||||
|
## Decisions locked
|
||||||
|
|
||||||
|
| Decision | Choice |
|
||||||
|
|---|---|
|
||||||
|
| Canonical source | `gitea.civilytics.org/Civilytics/uscogdata`, flipped public |
|
||||||
|
| Public mirror | `github.com/civilytics/uscogdata` — issues, PRs, multi-OS check, CDN |
|
||||||
|
| Mirror mechanism | Gitea Actions non-force `git push` (not a push mirror) |
|
||||||
|
| Binaries | `civilytics.r-universe.dev`, registry pinned to a release tag |
|
||||||
|
| Author of record | Jared E. Knowles `<jared@civilytics.com>`, ORCID `0000-0003-0005-9478` |
|
||||||
|
| Copyright | Civilytics Consulting LLC (`cph`, `fnd`) |
|
||||||
|
| License | MIT (package) · CC-BY-4.0 (corpus) |
|
||||||
|
| Corrections intake | Deferred — see *Out of scope* |
|
||||||
|
| Other packages | Parked until this one walks the path end to end |
|
||||||
|
|
||||||
|
## P0 — the corpus is unreachable
|
||||||
|
|
||||||
|
Two independent faults, either of which alone is fatal.
|
||||||
|
|
||||||
|
**No corpus URL exists.** `R/config.R` defaults to the literal
|
||||||
|
`REPLACE_WITH_SHARE_TOKEN` sentinel, and no file in the repo supplies a working
|
||||||
|
one. A new user calling any verb gets `uscogdata_url_not_configured` with no path
|
||||||
|
to resolution.
|
||||||
|
|
||||||
|
**Remote reads are broken regardless.** Every partitioned view globs:
|
||||||
|
|
||||||
|
```sql
|
||||||
|
FROM read_parquet('{url}data/long/**/*.parquet', hive_partitioning = true)
|
||||||
|
```
|
||||||
|
|
||||||
|
DuckDB 1.5.5 refuses globs over generic HTTP. Its suggested
|
||||||
|
`allow_asterisks_in_http_paths` escape hatch does not help — it forwards the
|
||||||
|
literal `**/*` as a filename and 404s, because plain HTTP exposes no directory
|
||||||
|
listing to expand against.
|
||||||
|
|
||||||
|
The package therefore works only against a **local path**. That is how the API
|
||||||
|
runs it (`CORPUS_HOST_PATH` is a host mount on maxwell) and how the tests run
|
||||||
|
(bundled fixture), which is why the fault went unnoticed. The README's headline
|
||||||
|
claim — *"Reads the published corpus directly from Nextcloud via DuckDB httpfs —
|
||||||
|
no local bulk downloads required"* — is currently false.
|
||||||
|
|
||||||
|
### Fix: enumerate from the manifest, do not glob
|
||||||
|
|
||||||
|
`manifest.json` already lists every partition under `files.long_partitions[]`
|
||||||
|
with `path`, `year`, `sha256`, `row_count` and `size_bytes` — 56 of them.
|
||||||
|
Substituting an explicit file list for the glob was measured against the
|
||||||
|
published corpus on 2026-08-08:
|
||||||
|
|
||||||
|
| Path | Result |
|
||||||
|
|---|---|
|
||||||
|
| `https://…/data/long/**/*.parquet` (default) | error — globs unsupported over HTTP |
|
||||||
|
| same, `allow_asterisks_in_http_paths = true` | error — literal `**/*` 404s |
|
||||||
|
| `hf://datasets/civilytics/us-cog-finance/…` glob | 46,148,034 rows |
|
||||||
|
| **explicit list over plain https** | **46,148,034 rows** |
|
||||||
|
|
||||||
|
`hive_partitioning = true` still recovers `year` from the paths under
|
||||||
|
enumeration, so no downstream view or verb changes.
|
||||||
|
|
||||||
|
Enumeration is preferred over `hf://` deliberately. It is **host-agnostic** —
|
||||||
|
Nextcloud, HuggingFace, or any static server take the same code path — where
|
||||||
|
`hf://` would tie the default to one vendor's protocol and still need
|
||||||
|
special-casing, since manifest fetching goes through `httr2`, which cannot speak
|
||||||
|
`hf://`. Enumeration also *removes* a dependency (globbing) rather than adding
|
||||||
|
one, and the manifest's per-file `sha256` becomes available for integrity
|
||||||
|
checking later.
|
||||||
|
|
||||||
|
Views are registered from `inst/sql/` with `{url}` substitution in
|
||||||
|
`R/views.R:.register_views()`. The list must be built once per session from the
|
||||||
|
already-fetched manifest and substituted the same way, so the change is confined
|
||||||
|
to view registration and does not touch verb code.
|
||||||
|
|
||||||
|
### Fix: ship a working default
|
||||||
|
|
||||||
|
`R/config.R`'s default becomes the public HuggingFace `resolve/main/` URL:
|
||||||
|
CC-BY-4.0, no token to publish, CDN-backed, and it keeps maxwell's uplink out of
|
||||||
|
the path — the same reasoning behind the GitHub mirror and r-universe.
|
||||||
|
|
||||||
|
This means `library(uscogdata)` followed by a verb works with **zero
|
||||||
|
configuration**, which is what makes the package demonstrable in a README and
|
||||||
|
later in a post. `USCOGDATA_URL` and `options(uscogdata.url=)` continue to
|
||||||
|
override, so the Nextcloud copy and local mirrors are unaffected.
|
||||||
|
|
||||||
|
The `uscogdata_url_not_configured` error class stays — it still fires for an
|
||||||
|
explicitly-set empty or placeholder URL — but ceases to be the default
|
||||||
|
experience.
|
||||||
|
|
||||||
|
### Consequence: `cog_mirror()` is promoted
|
||||||
|
|
||||||
|
Measured cost of the remote default, from efron on a good connection:
|
||||||
|
|
||||||
|
| | |
|
||||||
|
|---|---|
|
||||||
|
| Whole corpus | **190.6 MB**, 56 partitions, 46,148,034 rows, FY1967–FY2024 |
|
||||||
|
| One government, one year | 1.5 s |
|
||||||
|
| One government, all 56 years | 2.8 s |
|
||||||
|
| Disk written | **0.00 MB** — range requests only; `external_file_cache` is in-memory |
|
||||||
|
|
||||||
|
Nothing persists locally beyond the shared `httpfs` extension in `~/.duckdb` (a
|
||||||
|
few MB, once per machine, across all DuckDB use). Costs are RAM and per-query
|
||||||
|
bandwidth, since nothing caches between sessions.
|
||||||
|
|
||||||
|
Those timings are raw scans. Real verbs additionally join crosswalks, resolve
|
||||||
|
categories and assemble provenance, so end-to-end verb latency will be higher and
|
||||||
|
**must be re-measured once the fix lands** — it cannot be measured today.
|
||||||
|
|
||||||
|
The corpus being only 190.6 MB makes `cog_mirror()` a first-class option rather
|
||||||
|
than a developer footnote. The README presents **both paths**:
|
||||||
|
|
||||||
|
- **Remote (default, zero setup)** — trying it out, teaching, one-off questions.
|
||||||
|
- **Mirrored (`cog_mirror()`, 190 MB once)** — repeated or heavy analysis,
|
||||||
|
offline work, reproducibility, or preferring not to depend on HuggingFace.
|
||||||
|
|
||||||
|
The second is also the honest answer to the vendor-dependency question raised by
|
||||||
|
defaulting to HuggingFace: **the escape hatch is one function call and 190 MB**,
|
||||||
|
after which no analysis touches an external service. The README says so
|
||||||
|
explicitly. That is the difference between a convenience default and lock-in.
|
||||||
|
|
||||||
|
## Release-readiness fixes
|
||||||
|
|
||||||
|
| # | Issue | Fix |
|
||||||
|
|---|---|---|
|
||||||
|
| 1 | `MaxCorpusSchema: 5` in DESCRIPTION; `.validate_schema()` accepts `4,5,6,7`; published corpus is **7** | `MaxCorpusSchema: 7` |
|
||||||
|
| 2 | `^vignettes$` in `.Rbuildignore` — both vignettes absent from the installed package, while README tells users to run `vignette("total-spending")` | Remove `^vignettes$`, `^doc$`, `^Meta$`. Both vignettes build offline (`total-spending` reads the bundled fixture; `population-denominators` is `eval = FALSE`) |
|
||||||
|
| 3 | `_pkgdown.yml` reference index covers 6 of 14 exports — pkgdown errors on missing topics | Add `cog_categories`, `cog_explain`, `cog_find_peers`, `cog_geographic_rollup`, `cog_manifest`, `cog_mirror`, `cog_peer_compare`, `cog_recipes`; set `url:` |
|
||||||
|
| 4 | No `URL:` / `BugReports:` in DESCRIPTION | Add both, pointing at the GitHub mirror |
|
||||||
|
| 5 | No `LICENSE.md`; `LICENSE` holder reads `Civilytics` | `usethis::use_mit_license("Civilytics Consulting LLC")` |
|
||||||
|
| 6 | README instructs stripping the fixture at release | Delete that section — see below |
|
||||||
|
| 7 | `Authors@R` is an org with no human | Jared E. Knowles `aut`/`cre` + ORCID; Civilytics Consulting LLC `cph`/`fnd` |
|
||||||
|
|
||||||
|
**On #6.** The advice to add `^inst/extdata/fixture_corpus$` to `.Rbuildignore`
|
||||||
|
is CRAN-sized thinking (5 MB limit) and this package is not going to CRAN.
|
||||||
|
Stripping the 15 MB fixture would break `total-spending.Rmd`, which reads from
|
||||||
|
it, and would leave r-universe and GitHub Actions unable to run the 28 test files
|
||||||
|
without a corpus credential. **The fixture is what lets `R CMD check` pass
|
||||||
|
anywhere with zero secrets** — precisely what public CI needs. It ships.
|
||||||
|
|
||||||
|
## README
|
||||||
|
|
||||||
|
The current README addresses someone standing inside the repo tree: status reads
|
||||||
|
"Under active development (Phase 2 of the cog_pipeline project)", it points at
|
||||||
|
`../cog_pipeline/docs/reader-specification.md`, the install line is commented
|
||||||
|
out, and developer, testing and release sections sit above anything a user needs.
|
||||||
|
|
||||||
|
Restructured around a stranger, in this order:
|
||||||
|
|
||||||
|
1. **What this is** — one paragraph, and what the corpus covers (types 0–3,
|
||||||
|
FY1967–FY2024, 46M rows, 190.6 MB).
|
||||||
|
2. **Install** — r-universe first (binaries), git second.
|
||||||
|
3. **Quickstart that actually runs** — resolve a government, get its history,
|
||||||
|
print provenance. No configuration step.
|
||||||
|
4. **Two ways to read the corpus** — remote default vs `cog_mirror()`, with the
|
||||||
|
measured numbers and the independence note.
|
||||||
|
5. **Amounts are in full US dollars** — kept near the top. This is the errata
|
||||||
|
most likely to produce a wrong answer that looks plausible.
|
||||||
|
6. **Concepts** — primary/direct/total spending, general/total revenue,
|
||||||
|
coverage. Condensed, linking to the vignettes for the full treatment.
|
||||||
|
7. **How to cite** — `citation("uscogdata")`, corpus CC-BY-4.0 attribution.
|
||||||
|
8. **Contributing** — canonical-on-Gitea PR flow.
|
||||||
|
|
||||||
|
Developer notes, testing instructions and release procedure move to
|
||||||
|
`CONTRIBUTING.md`. Every path reference to a sibling repo is removed or replaced
|
||||||
|
with a URL that resolves for someone who has only this repo.
|
||||||
|
|
||||||
|
## NEWS.md
|
||||||
|
|
||||||
|
`NEWS.md` currently holds two sections. `0.2.0` is a legitimate changelog — the
|
||||||
|
`"All Categories"` reserved value, the coverage-signposting fix, the
|
||||||
|
`n_units_reporting` documentation — and it stays. Beneath it,
|
||||||
|
`0.1.0 (development)` is a pre-release churn log: changes described relative to
|
||||||
|
states no user has ever seen ("Breaking: corpus schema_version 4", "the package
|
||||||
|
now requires…"), spanning the package's entire pre-release development. To a
|
||||||
|
newcomer deciding whether to depend on this, that section reads as instability.
|
||||||
|
|
||||||
|
**A new `0.3.0` section is added at the top, framed as the first public
|
||||||
|
release**: what the package does, what the corpus covers, and the caveats that
|
||||||
|
are genuinely load-bearing. **`0.2.0` is kept verbatim.** **`0.1.0 (development)`
|
||||||
|
is dropped** — that history stays in git, where it belongs.
|
||||||
|
|
||||||
|
The version is `0.3.0` rather than `0.2.0` because this release changes
|
||||||
|
user-visible behaviour: remote corpus reads go from broken to working, and the
|
||||||
|
default URL from a dead placeholder to a live corpus. It is also not `1.0.0` —
|
||||||
|
the corpus still excludes government types 4 and 5 pending validation, so a
|
||||||
|
stability promise would overclaim. No git tag exists for any prior version;
|
||||||
|
`chore: release 0.2.0` bumped `DESCRIPTION` and `NEWS` only.
|
||||||
|
|
||||||
|
The substantive content is migrated, not deleted. These are hard-won and belong
|
||||||
|
in documentation rather than buried in a changelog:
|
||||||
|
|
||||||
|
| Content | Destination |
|
||||||
|
|---|---|
|
||||||
|
| Coverage disclosure on multi-government aggregates (census vs sample years) | README concepts + `cog_geographic_rollup()` docs |
|
||||||
|
| `complete = TRUE` three-way absence semantics (`reported` / `census_zero` / `not_reported`) | `cog_spending()` / `cog_revenue()` docs |
|
||||||
|
| Series-break and corpus-break surfacing | README + `cog_explain()` docs |
|
||||||
|
| $1,000s → full dollars conversion | README, already prominent |
|
||||||
|
| Per-year F-33 population denominators | `population-denominators` vignette, already there |
|
||||||
|
|
||||||
|
This also makes NEWS reusable as raw material for the release announcement,
|
||||||
|
which is the stated downstream purpose.
|
||||||
|
|
||||||
|
## Distribution mechanics
|
||||||
|
|
||||||
|
Recorded as decided; executed after the package is clean and checks are green.
|
||||||
|
|
||||||
|
**Sequence matters.** r-universe publishes check results the moment a package is
|
||||||
|
registered. Registering before the fixes above land means a red badge on day one,
|
||||||
|
which is a worse first impression than a week's delay.
|
||||||
|
|
||||||
|
1. `gitleaks` over full history. A coarse grep found nothing across 140 commits
|
||||||
|
and the default corpus URL is still the placeholder sentinel, but a proper
|
||||||
|
scan is the gate on an irreversible action.
|
||||||
|
2. Flip the Gitea repo public. Disable Gitea issues on it, so there is exactly
|
||||||
|
one inbox.
|
||||||
|
3. Create `github.com/civilytics/uscogdata`. Add `.github/workflows/` for the
|
||||||
|
Windows/macOS/Linux `R CMD check` matrix — the platforms the Gitea runner
|
||||||
|
cannot provide, and which this package has never been tested on despite
|
||||||
|
depending on duckdb and httr2. Gitea reads `.gitea/workflows`, GitHub reads
|
||||||
|
`.github/workflows`; both live in one tree without colliding.
|
||||||
|
4. Gitea Actions workflow pushing to GitHub **without `--force`**, so divergence
|
||||||
|
fails loudly in CI rather than silently overwriting.
|
||||||
|
5. Add `jared@civilytics.com` as a verified secondary email on the GitHub
|
||||||
|
account — r-universe links maintainer identity by matching DESCRIPTION's email
|
||||||
|
against registered GitHub emails, and the association only takes effect on the
|
||||||
|
next build.
|
||||||
|
6. Tag `v0.3.0`. Create `github.com/civilytics/civilytics.r-universe.dev` with a
|
||||||
|
`packages.json` pinned to the tag, pointing at the GitHub mirror rather than
|
||||||
|
Gitea so clone traffic stays off maxwell. Install the r-universe app.
|
||||||
|
|
||||||
|
### PR flow
|
||||||
|
|
||||||
|
Never press Merge on GitHub. A merge there is overwritten by the next sync, the
|
||||||
|
PR still displays "Merged", and nothing says otherwise.
|
||||||
|
|
||||||
|
```sh
|
||||||
|
git remote add github https://github.com/civilytics/uscogdata.git
|
||||||
|
git config --add remote.github.fetch '+refs/pull/*/head:refs/remotes/github/pr/*'
|
||||||
|
git fetch github
|
||||||
|
git switch -c pr-42 github/pr/42 # test
|
||||||
|
git switch main && git merge --no-ff pr-42
|
||||||
|
git push origin main # Gitea -> mirror -> GitHub
|
||||||
|
```
|
||||||
|
|
||||||
|
GitHub auto-closes a PR as merged once its head commit becomes an ancestor of the
|
||||||
|
base branch, so `--no-ff` — which preserves the contributor's SHAs — makes the PR
|
||||||
|
close itself when the mirror pushes. **For external PRs, merge; do not squash or
|
||||||
|
rebase.** Squashing rewrites the SHAs, the auto-close never fires, and closing by
|
||||||
|
hand reads to a first-time contributor as rejection.
|
||||||
|
|
||||||
|
`CONTRIBUTING.md` states this, and a GitHub Action comments it on incoming PRs.
|
||||||
|
No CLA; no DCO.
|
||||||
|
|
||||||
|
## Verification
|
||||||
|
|
||||||
|
The release is not done until all of these pass:
|
||||||
|
|
||||||
|
1. `R CMD check --as-cran` clean on Linux, and on Windows and macOS via the
|
||||||
|
GitHub matrix. This package has never been checked on the latter two.
|
||||||
|
2. Full test suite (28 files) green against the **bundled fixture**, offline,
|
||||||
|
with no credentials — the property public CI depends on.
|
||||||
|
3. Full test suite green against the **live corpus**, which additionally
|
||||||
|
exercises the enumeration fix that the fixture's local path cannot.
|
||||||
|
4. `pkgdown::build_site()` completes.
|
||||||
|
5. Both vignettes present in the built tarball and
|
||||||
|
`vignette("total-spending", package = "uscogdata")` resolves from an
|
||||||
|
installed copy.
|
||||||
|
6. **Cold-start check on a machine that has never seen this package:** install
|
||||||
|
from r-universe, `library(uscogdata)`, run the README quickstart verbatim with
|
||||||
|
no environment variables set. This is the only test that catches the P0 class
|
||||||
|
of fault, and its absence is why the fault survived.
|
||||||
|
7. End-to-end verb latency re-measured against the live corpus and the README's
|
||||||
|
numbers updated if they moved.
|
||||||
|
|
||||||
|
## Out of scope
|
||||||
|
|
||||||
|
- **Corrections intake.** Deferred by decision. Consequence: the release cannot
|
||||||
|
invite data-error reports or make the "traceable and correctable" claim that
|
||||||
|
most distinguishes this corpus from Census's own files. `BugReports:` points at
|
||||||
|
package issues only. A verified correction should eventually terminate as a
|
||||||
|
`lineage_event` or `series_break` row so it propagates through provenance to
|
||||||
|
every consumer — that design is unstarted.
|
||||||
|
- **Announcement posts.** Deferred. The API announcement is gated on corrections
|
||||||
|
landing and merits a Civic Pulse edition.
|
||||||
|
- **The rest of the R package backlog.** Parked until this one completes the path.
|
||||||
|
- **`cog_pipeline` publication.** Stays private.
|
||||||
@@ -6,6 +6,33 @@ fixture_corpus_path <- function() {
|
|||||||
if (nzchar(p)) paste0(p, "/") else ""
|
if (nzchar(p)) paste0(p, "/") else ""
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# Path to a file in the SOURCE tree (README.md, man/*.Rd, vignettes/*.Rmd),
|
||||||
|
# or "" when it isn't there.
|
||||||
|
#
|
||||||
|
# Tests that assert on documentation content have to read the sources, and the
|
||||||
|
# sources only exist when the suite runs from a checkout. Under R CMD check the
|
||||||
|
# suite runs from the INSTALLED package, where man/ and vignettes/ are not
|
||||||
|
# shipped and `../../README.md` does not resolve -- so those tests must skip
|
||||||
|
# rather than error. CI runs testthat::test_local() from the checkout BEFORE
|
||||||
|
# rcmdcheck, so the assertions are still enforced on every push; this only
|
||||||
|
# stops them from failing a context that structurally cannot satisfy them.
|
||||||
|
source_tree_path <- function(...) {
|
||||||
|
p <- testthat::test_path("..", "..", ...)
|
||||||
|
if (file.exists(p)) p else ""
|
||||||
|
}
|
||||||
|
|
||||||
|
# Skip unless every named source file is present (see source_tree_path()).
|
||||||
|
skip_if_no_source_tree <- function(...) {
|
||||||
|
paths <- vapply(list(...), function(rel) do.call(source_tree_path, as.list(rel)),
|
||||||
|
character(1))
|
||||||
|
missing <- vapply(paths, function(p) !nzchar(p), logical(1))
|
||||||
|
testthat::skip_if(
|
||||||
|
any(missing),
|
||||||
|
"package source tree not available (running against the installed package)"
|
||||||
|
)
|
||||||
|
invisible(paths)
|
||||||
|
}
|
||||||
|
|
||||||
# Skip a test if no corpus is reachable (bundled fixture or explicit remote URL).
|
# Skip a test if no corpus is reachable (bundled fixture or explicit remote URL).
|
||||||
skip_if_no_corpus <- function() {
|
skip_if_no_corpus <- function() {
|
||||||
p <- fixture_corpus_path()
|
p <- fixture_corpus_path()
|
||||||
@@ -26,3 +53,137 @@ with_fixture_corpus <- function(code) {
|
|||||||
}, add = TRUE)
|
}, add = TRUE)
|
||||||
force(code)
|
force(code)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# Copy the bundled fixture to a temp dir with manifest.json's schema_version
|
||||||
|
# patched to `version`, then run `code` against it with a clean session
|
||||||
|
# (mirrors with_fixture_corpus()). Used to exercise the v4/v5 dual-accept
|
||||||
|
# path without a second physical fixture tree: a real v4 corpus has no
|
||||||
|
# harmonization_map/harmonization_recipes/series_breaks parquet files, but
|
||||||
|
# .register_views() only *reads* those when schema_version >= 5 (see
|
||||||
|
# R/views.R), so a doctored copy of the (v5) bundled fixture with the
|
||||||
|
# manifest's schema_version knocked down to 4 is a faithful stand-in.
|
||||||
|
with_doctored_schema_version <- function(version, code) {
|
||||||
|
src <- fixture_corpus_path()
|
||||||
|
tmp <- withr::local_tempdir(.local_envir = parent.frame())
|
||||||
|
file.copy(list.files(src, full.names = TRUE), tmp, recursive = TRUE)
|
||||||
|
|
||||||
|
manifest_path <- file.path(tmp, "manifest.json")
|
||||||
|
m <- jsonlite::fromJSON(manifest_path, simplifyVector = FALSE)
|
||||||
|
m$schema_version <- as.integer(version)
|
||||||
|
writeLines(
|
||||||
|
jsonlite::toJSON(m, auto_unbox = TRUE, pretty = TRUE, null = "null"),
|
||||||
|
manifest_path
|
||||||
|
)
|
||||||
|
|
||||||
|
old_url <- Sys.getenv("USCOGDATA_URL", unset = NA)
|
||||||
|
uscogdata:::cog_close()
|
||||||
|
Sys.setenv(USCOGDATA_URL = paste0(tmp, "/"))
|
||||||
|
on.exit({
|
||||||
|
uscogdata:::cog_close()
|
||||||
|
if (is.na(old_url)) Sys.unsetenv("USCOGDATA_URL") else Sys.setenv(USCOGDATA_URL = old_url)
|
||||||
|
}, add = TRUE)
|
||||||
|
force(code)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Copy the bundled fixture to a temp dir with representation.parquet and
|
||||||
|
# code_set.parquet removed (and dropped from the manifest's metadata list),
|
||||||
|
# then run `code` against it. Models a corpus published BEFORE sparsification:
|
||||||
|
# schema_version is left alone deliberately, because it was never bumped for
|
||||||
|
# that change -- the pre-sparsification fixture this package shipped until
|
||||||
|
# 2026-07-30 was schema v6 and carried neither table. Presence in the manifest
|
||||||
|
# is therefore the only honest signal, and this helper is what proves the
|
||||||
|
# package keys off it rather than off the version number.
|
||||||
|
with_corpus_missing_representation <- function(code) {
|
||||||
|
src <- fixture_corpus_path()
|
||||||
|
tmp <- withr::local_tempdir(.local_envir = parent.frame())
|
||||||
|
file.copy(list.files(src, full.names = TRUE), tmp, recursive = TRUE)
|
||||||
|
|
||||||
|
dropped <- c("representation.parquet", "code_set.parquet")
|
||||||
|
file.remove(file.path(tmp, "data", dropped))
|
||||||
|
|
||||||
|
manifest_path <- file.path(tmp, "manifest.json")
|
||||||
|
m <- jsonlite::fromJSON(manifest_path, simplifyVector = FALSE)
|
||||||
|
m$files$metadata <- Filter(
|
||||||
|
function(f) !basename(f$path) %in% dropped, m$files$metadata
|
||||||
|
)
|
||||||
|
writeLines(
|
||||||
|
jsonlite::toJSON(m, auto_unbox = TRUE, pretty = TRUE, null = "null"),
|
||||||
|
manifest_path
|
||||||
|
)
|
||||||
|
|
||||||
|
old_url <- Sys.getenv("USCOGDATA_URL", unset = NA)
|
||||||
|
uscogdata:::cog_close()
|
||||||
|
Sys.setenv(USCOGDATA_URL = paste0(tmp, "/"))
|
||||||
|
on.exit({
|
||||||
|
uscogdata:::cog_close()
|
||||||
|
if (is.na(old_url)) Sys.unsetenv("USCOGDATA_URL") else Sys.setenv(USCOGDATA_URL = old_url)
|
||||||
|
}, add = TRUE)
|
||||||
|
force(code)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Copy the bundled fixture to a temp dir with summary_categories.parquet
|
||||||
|
# rewritten to drop every M/L (intergovernmental) row, then run `code`
|
||||||
|
# against it with a clean session (mirrors with_fixture_corpus()/
|
||||||
|
# with_doctored_schema_version()). Models a real pre-cog_pipeline-PR#59
|
||||||
|
# corpus: the 66 M/L category rows shipped with NO schema_version bump (see
|
||||||
|
# C2 in the expenditure-concept review), so schema_version is left
|
||||||
|
# untouched here -- only the category data itself is rolled back.
|
||||||
|
with_corpus_missing_ig_categories <- function(code) {
|
||||||
|
src <- fixture_corpus_path()
|
||||||
|
tmp <- withr::local_tempdir(.local_envir = parent.frame())
|
||||||
|
file.copy(list.files(src, full.names = TRUE), tmp, recursive = TRUE)
|
||||||
|
|
||||||
|
cats_path <- file.path(tmp, "data", "summary_categories.parquet")
|
||||||
|
filtered_path <- file.path(tmp, "data", "summary_categories_filtered.parquet")
|
||||||
|
write_con <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
on.exit(DBI::dbDisconnect(write_con, shutdown = TRUE), add = TRUE)
|
||||||
|
DBI::dbExecute(write_con, sprintf(
|
||||||
|
"COPY (SELECT * FROM read_parquet(%s) WHERE LEFT(item_code, 1) NOT IN ('M', 'L'))
|
||||||
|
TO %s (FORMAT PARQUET)",
|
||||||
|
uscogdata:::.sql_lit_chr(cats_path), uscogdata:::.sql_lit_chr(filtered_path)
|
||||||
|
))
|
||||||
|
file.remove(cats_path)
|
||||||
|
file.rename(filtered_path, cats_path)
|
||||||
|
|
||||||
|
old_url <- Sys.getenv("USCOGDATA_URL", unset = NA)
|
||||||
|
uscogdata:::cog_close()
|
||||||
|
Sys.setenv(USCOGDATA_URL = paste0(tmp, "/"))
|
||||||
|
on.exit({
|
||||||
|
uscogdata:::cog_close()
|
||||||
|
if (is.na(old_url)) Sys.unsetenv("USCOGDATA_URL") else Sys.setenv(USCOGDATA_URL = old_url)
|
||||||
|
}, add = TRUE)
|
||||||
|
force(code)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Copy the bundled fixture to a temp dir with summary_categories.parquet
|
||||||
|
# rewritten to DROP the balance_subtype column, then run `code` against it.
|
||||||
|
# Models a corpus published before cog_pipeline #76/#77. schema_version is
|
||||||
|
# left untouched deliberately: that change shipped without a version bump, so
|
||||||
|
# column presence is the only honest signal -- this helper is what proves the
|
||||||
|
# package keys off it. Mirrors with_corpus_missing_ig_categories().
|
||||||
|
with_corpus_missing_balance_subtype <- function(code) {
|
||||||
|
src <- fixture_corpus_path()
|
||||||
|
tmp <- withr::local_tempdir(.local_envir = parent.frame())
|
||||||
|
file.copy(list.files(src, full.names = TRUE), tmp, recursive = TRUE)
|
||||||
|
|
||||||
|
cats_path <- file.path(tmp, "data", "summary_categories.parquet")
|
||||||
|
filtered_path <- file.path(tmp, "data", "summary_categories_filtered.parquet")
|
||||||
|
write_con <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
on.exit(DBI::dbDisconnect(write_con, shutdown = TRUE), add = TRUE)
|
||||||
|
DBI::dbExecute(write_con, sprintf(
|
||||||
|
"COPY (SELECT * EXCLUDE (balance_subtype) FROM read_parquet(%s))
|
||||||
|
TO %s (FORMAT PARQUET)",
|
||||||
|
uscogdata:::.sql_lit_chr(cats_path), uscogdata:::.sql_lit_chr(filtered_path)
|
||||||
|
))
|
||||||
|
file.remove(cats_path)
|
||||||
|
file.rename(filtered_path, cats_path)
|
||||||
|
|
||||||
|
old_url <- Sys.getenv("USCOGDATA_URL", unset = NA)
|
||||||
|
uscogdata:::cog_close()
|
||||||
|
Sys.setenv(USCOGDATA_URL = paste0(tmp, "/"))
|
||||||
|
on.exit({
|
||||||
|
uscogdata:::cog_close()
|
||||||
|
if (is.na(old_url)) Sys.unsetenv("USCOGDATA_URL") else Sys.setenv(USCOGDATA_URL = old_url)
|
||||||
|
}, add = TRUE)
|
||||||
|
force(code)
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,44 @@
|
|||||||
|
# Helper for the Madison-walkthrough finding tests (uscogdata #11-#16).
|
||||||
|
#
|
||||||
|
# Those tests all assert something about what a `cog_*` verb includes or
|
||||||
|
# excludes. The expected amounts must therefore come from the RAW corpus, never
|
||||||
|
# from the verb under test: verifying an absence through the filter that creates
|
||||||
|
# it proves nothing. `wt_raw_*()` opens its own DuckDB connection straight onto
|
||||||
|
# the corpus's `long` parquet partitions, bypassing uscogdata's SQL views (and
|
||||||
|
# therefore its `flow_prefixes` filtering) entirely.
|
||||||
|
|
||||||
|
wt_corpus_glob <- function() {
|
||||||
|
url <- Sys.getenv("USCOGDATA_URL")
|
||||||
|
if (!nzchar(url)) testthat::skip("USCOGDATA_URL is not set")
|
||||||
|
paste0(sub("/$", "", url), "/data/long/**/*.parquet")
|
||||||
|
}
|
||||||
|
|
||||||
|
wt_raw_query <- function(sql) {
|
||||||
|
con <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE)
|
||||||
|
DBI::dbGetQuery(con, sql)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Sum of `amt` (in $1,000s, as the corpus stores it) for one government-year,
|
||||||
|
# restricted either to an explicit set of item codes or to a set of first-letter
|
||||||
|
# prefixes. Aggregate rows are excluded, matching every published verb.
|
||||||
|
wt_raw_amt <- function(govid, year, codes = NULL, prefixes = NULL) {
|
||||||
|
stopifnot(xor(is.null(codes), is.null(prefixes)))
|
||||||
|
filter_sql <- if (!is.null(codes)) {
|
||||||
|
paste0("item_code IN (", paste0("'", codes, "'", collapse = ", "), ")")
|
||||||
|
} else {
|
||||||
|
paste0("LEFT(item_code, 1) IN (", paste0("'", prefixes, "'", collapse = ", "), ")")
|
||||||
|
}
|
||||||
|
out <- wt_raw_query(paste0(
|
||||||
|
"SELECT COALESCE(SUM(amt), 0) AS amt FROM read_parquet('", wt_corpus_glob(), "') ",
|
||||||
|
"WHERE canonical_govid = '", govid, "' AND year = ", year,
|
||||||
|
" AND NOT is_aggregate AND ", filter_sql
|
||||||
|
))
|
||||||
|
out$amt[[1]]
|
||||||
|
}
|
||||||
|
|
||||||
|
# The item codes a verb reports having summed, flattened out of the
|
||||||
|
# comma-separated `codes_included` column.
|
||||||
|
wt_codes_included <- function(df) {
|
||||||
|
sort(unique(trimws(unlist(strsplit(stats::na.omit(df$codes_included), ",")))))
|
||||||
|
}
|
||||||
@@ -41,3 +41,10 @@ test_that(".inflate preserves NA amounts", {
|
|||||||
expect_true(is.na(result[2]))
|
expect_true(is.na(result[2]))
|
||||||
expect_false(any(is.na(result[c(1, 3)])))
|
expect_false(any(is.na(result[c(1, 3)])))
|
||||||
})
|
})
|
||||||
|
|
||||||
|
test_that("bundled CPI covers the full 1967+ corpus era through this year", {
|
||||||
|
cpi <- .cpi_table()
|
||||||
|
expect_lte(min(cpi$year), 1967L)
|
||||||
|
expect_gte(max(cpi$year), as.integer(format(Sys.Date(), "%Y")))
|
||||||
|
expect_false(any(is.na(cpi$cpi)))
|
||||||
|
})
|
||||||
|
|||||||
@@ -0,0 +1,58 @@
|
|||||||
|
test_that('cog_geographic_rollup() accepts "All Categories" and agrees with per-category sums', {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
govs <- cog_gov_search(name = NULL, state = "WI", type = 2L)
|
||||||
|
expect_gt(nrow(govs), 1L)
|
||||||
|
ids <- list(city = utils::head(govs$canonical_govid, 25L))
|
||||||
|
|
||||||
|
by_cat <- cog_geographic_rollup(ids, category = NULL, years = 2019L)
|
||||||
|
total <- cog_geographic_rollup(ids, category = "All Categories", years = 2019L)
|
||||||
|
|
||||||
|
expect_setequal(unique(total$category), "All Categories")
|
||||||
|
# one row per (govid, subtype) that appears in the per-category result
|
||||||
|
key_by_cat <- unique(paste(by_cat$canonical_govid, by_cat$spend_subtype))
|
||||||
|
key_total <- paste(total$canonical_govid, total$spend_subtype)
|
||||||
|
expect_setequal(key_total, key_by_cat)
|
||||||
|
|
||||||
|
lhs <- tapply(by_cat$amt_nominal, paste(by_cat$canonical_govid, by_cat$spend_subtype), sum)
|
||||||
|
rhs <- tapply(total$amt_nominal, key_total, sum)
|
||||||
|
expect_equal(as.numeric(rhs[names(lhs)]), as.numeric(lhs), tolerance = 1e-8)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('"All Categories" survives per_capita and inflation adjustment through the rollup', {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
govs <- cog_gov_search(name = NULL, state = "WI", type = 2L)
|
||||||
|
ids <- list(city = utils::head(govs$canonical_govid, 10L))
|
||||||
|
r <- cog_geographic_rollup(ids, category = "All Categories", years = 2019L,
|
||||||
|
per_capita = TRUE, adjust_to_year = 2020L)
|
||||||
|
expect_true(all(c("amt_per_capita_nominal", "amt_real", "amt_per_capita_real") %in% names(r)))
|
||||||
|
expect_setequal(unique(r$category), "All Categories")
|
||||||
|
expect_true(all(is.finite(r$amt_real)))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('cog_geographic_rollup() still refuses expenditure_concept = "total" with "All Categories"', {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
govs <- cog_gov_search(name = NULL, state = "WI", type = 2L)
|
||||||
|
ids <- list(city = utils::head(govs$canonical_govid, 5L))
|
||||||
|
expect_error(
|
||||||
|
cog_geographic_rollup(ids, category = "All Categories", years = 2019L,
|
||||||
|
expenditure_concept = "total")
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("n_units_reporting is category-conditional, not a response rate", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
govs <- cog_gov_search(name = NULL, state = "WI", type = 2L)
|
||||||
|
ids <- list(city = govs$canonical_govid)
|
||||||
|
|
||||||
|
police <- cog_geographic_rollup(ids, category = "Police", years = 2012L)
|
||||||
|
allcat <- cog_geographic_rollup(ids, category = "All Categories", years = 2012L)
|
||||||
|
|
||||||
|
cov_police <- cog_explain(police, format = "list")$coverage
|
||||||
|
cov_all <- cog_explain(allcat, format = "list")$coverage
|
||||||
|
|
||||||
|
# Same year, same requested govids, same collection -- yet a single category
|
||||||
|
# reports fewer units than the all-categories query. That gap is real zeros,
|
||||||
|
# not non-response, which is exactly why the ratio is not a response rate.
|
||||||
|
expect_lte(cov_police$n_units_reporting, cov_all$n_units_reporting)
|
||||||
|
expect_identical(cov_police$n_units_expected, cov_all$n_units_expected)
|
||||||
|
})
|
||||||
@@ -0,0 +1,267 @@
|
|||||||
|
# Baseline at branch point: 843 PASS / 0 FAIL / 0 SKIP / 0 WARN (2026-08-05, origin/main 2fc9e75)
|
||||||
|
|
||||||
|
test_that(".build_verb_sql emits a literal category and no category filter in all-categories mode", {
|
||||||
|
sql <- uscogdata:::.build_verb_sql(
|
||||||
|
view = "spending_annotated",
|
||||||
|
subtype_col = "spend_subtype",
|
||||||
|
cohort = uscogdata:::.make_cohort("552025209777"),
|
||||||
|
years = 2019L,
|
||||||
|
category = NULL,
|
||||||
|
subtype_scope = c("operations", "capital"),
|
||||||
|
all_categories = TRUE
|
||||||
|
)
|
||||||
|
|
||||||
|
expect_match(sql, "'All Categories' AS category", fixed = TRUE)
|
||||||
|
# no category filter of any kind
|
||||||
|
expect_false(grepl("AND category IN", sql, fixed = TRUE))
|
||||||
|
# category is not a grouping key
|
||||||
|
expect_false(grepl("GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, spend_subtype, category",
|
||||||
|
sql, fixed = TRUE))
|
||||||
|
# the subtype allowlist still applies -- this is what makes the sum a concept
|
||||||
|
expect_match(sql, "AND spend_subtype IN ('operations','capital')", fixed = TRUE)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".build_verb_sql is unchanged when all_categories is FALSE", {
|
||||||
|
args <- list(
|
||||||
|
view = "spending_annotated", subtype_col = "spend_subtype",
|
||||||
|
cohort = uscogdata:::.make_cohort("552025209777"), years = 2019L, category = NULL,
|
||||||
|
subtype_scope = c("operations", "capital")
|
||||||
|
)
|
||||||
|
old <- do.call(uscogdata:::.build_verb_sql, args)
|
||||||
|
new <- do.call(uscogdata:::.build_verb_sql, c(args, list(all_categories = FALSE)))
|
||||||
|
expect_identical(old, new)
|
||||||
|
expect_match(new, "GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, spend_subtype, category",
|
||||||
|
fixed = TRUE)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".ALL_CATEGORIES is the exact reserved string", {
|
||||||
|
expect_identical(uscogdata:::.ALL_CATEGORIES, "All Categories")
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('cog_spending(category = "All Categories") sums to the per-category total', {
|
||||||
|
gov <- "552025209777"
|
||||||
|
by_cat <- cog_spending(gov, 2019L)
|
||||||
|
total <- cog_spending(gov, 2019L, category = "All Categories")
|
||||||
|
|
||||||
|
expect_true(nrow(total) > 0L)
|
||||||
|
expect_setequal(unique(total$category), "All Categories")
|
||||||
|
# one row per subtype present in the by-category result
|
||||||
|
expect_setequal(unique(total$spend_subtype), unique(by_cat$spend_subtype))
|
||||||
|
expect_equal(nrow(total), length(unique(by_cat$spend_subtype)))
|
||||||
|
|
||||||
|
# the dollars agree, per subtype
|
||||||
|
lhs <- tapply(by_cat$amt_nominal, by_cat$spend_subtype, sum)
|
||||||
|
rhs <- tapply(total$amt_nominal, total$spend_subtype, sum)
|
||||||
|
expect_equal(as.numeric(rhs[names(lhs)]), as.numeric(lhs), tolerance = 1e-8)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('"All Categories" respects expenditure_concept', {
|
||||||
|
gov <- "552025209777"
|
||||||
|
prim <- cog_spending(gov, 2019L, category = "All Categories",
|
||||||
|
expenditure_concept = "primary")
|
||||||
|
dir <- cog_spending(gov, 2019L, category = "All Categories",
|
||||||
|
expenditure_concept = "direct")
|
||||||
|
# direct = primary plus interest and insurance benefits, so it is never smaller
|
||||||
|
expect_gte(sum(dir$amt_nominal), sum(prim$amt_nominal))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('"All Categories" works on revenue and respects revenue_concept', {
|
||||||
|
gov <- "552025209777"
|
||||||
|
gen <- cog_revenue(gov, 2019L, category = "All Categories",
|
||||||
|
revenue_concept = "general")
|
||||||
|
tot <- cog_revenue(gov, 2019L, category = "All Categories",
|
||||||
|
revenue_concept = "total")
|
||||||
|
expect_setequal(unique(gen$category), "All Categories")
|
||||||
|
expect_gte(sum(tot$amt_nominal), sum(gen$amt_nominal))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('"All Categories" cannot be combined with another category', {
|
||||||
|
expect_error(
|
||||||
|
cog_spending("552025209777", 2019L, category = c("All Categories", "Police")),
|
||||||
|
class = "uscogdata_all_categories_not_combinable"
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('"All Categories" is recorded in provenance', {
|
||||||
|
r <- cog_spending("552025209777", 2019L, category = "All Categories")
|
||||||
|
expect_identical(cog_explain(r, format = "list")$category, "All Categories")
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('"All Categories" combines with subtype to give operating totals', {
|
||||||
|
gov <- "552025209777"
|
||||||
|
ops_by_cat <- cog_spending(gov, 2019L)
|
||||||
|
ops_by_cat <- ops_by_cat[ops_by_cat$spend_subtype == "operations", ]
|
||||||
|
ops_total <- cog_spending(gov, 2019L, category = "All Categories")
|
||||||
|
ops_total <- ops_total[ops_total$spend_subtype == "operations", ]
|
||||||
|
expect_equal(sum(ops_total$amt_nominal), sum(ops_by_cat$amt_nominal),
|
||||||
|
tolerance = 1e-8)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('cog_categories() advertises "All Categories" for both flows', {
|
||||||
|
all <- cog_categories()
|
||||||
|
rows <- all[all$category == "All Categories", ]
|
||||||
|
expect_setequal(rows$category_type, c("expenditure", "revenue"))
|
||||||
|
expect_true(all(is.na(rows$subtype)))
|
||||||
|
expect_true(all(is.na(rows$n_codes)))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('cog_categories(type=) still scopes, including the pseudo-category', {
|
||||||
|
sp <- cog_categories(type = "spending")
|
||||||
|
expect_setequal(unique(sp$category_type), "expenditure")
|
||||||
|
expect_true("All Categories" %in% sp$category)
|
||||||
|
|
||||||
|
rev <- cog_categories(type = "revenue")
|
||||||
|
expect_setequal(unique(rev$category_type), "revenue")
|
||||||
|
expect_true("All Categories" %in% rev$category)
|
||||||
|
|
||||||
|
# balances have no concept vocabulary, so no pseudo-category
|
||||||
|
bal <- cog_categories(type = "balance")
|
||||||
|
expect_false("All Categories" %in% bal$category)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('cog_categories(pattern=) matches the pseudo-category', {
|
||||||
|
hit <- cog_categories(pattern = "^All Categories$")
|
||||||
|
expect_equal(nrow(hit), 2L)
|
||||||
|
})
|
||||||
|
|
||||||
|
# --- final whole-branch review fixes ---------------------------------------
|
||||||
|
|
||||||
|
test_that('complete = TRUE is refused when combined with "All Categories"', {
|
||||||
|
# .completion_grid_sql() would emit `AND c.category IN ('All Categories')`,
|
||||||
|
# match zero crosswalk rows, and the early return in .complete_result()
|
||||||
|
# would stamp completion$applied = TRUE, rows_filled = 0 -- reading as "the
|
||||||
|
# grid was checked and nothing was missing" when nothing was actually
|
||||||
|
# checked. Filling a summed row has no defined semantics, so the verb must
|
||||||
|
# refuse the combination outright (finding 2).
|
||||||
|
expect_error(
|
||||||
|
cog_spending("552025209777", 2019L, category = "All Categories",
|
||||||
|
complete = TRUE),
|
||||||
|
class = "uscogdata_complete_unsupported"
|
||||||
|
)
|
||||||
|
expect_error(
|
||||||
|
cog_revenue("552025209777", 2019L, category = "All Categories",
|
||||||
|
complete = TRUE),
|
||||||
|
class = "uscogdata_complete_unsupported"
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('cog_balances() rejects "All Categories" instead of silently returning zero rows', {
|
||||||
|
# cog_balances() reuses .validate_verb_inputs() but did not pass
|
||||||
|
# allow_all_categories = TRUE, so "All Categories" used to become
|
||||||
|
# `AND category IN ('All Categories')` against balance_annotated -- 0
|
||||||
|
# matching crosswalk rows, 0 rows back, no error (finding 3). Holdings are
|
||||||
|
# a stock with no concept vocabulary to sum across, so the honest answer is
|
||||||
|
# to refuse, the same way cog_spending()/cog_revenue() refuse other
|
||||||
|
# nonsensical combinations.
|
||||||
|
expect_error(
|
||||||
|
cog_balances("552025209777", 2019L, category = "All Categories"),
|
||||||
|
class = "uscogdata_all_categories_unsupported"
|
||||||
|
)
|
||||||
|
# An ordinary category still works -- this is not a blanket regression.
|
||||||
|
r <- suppressMessages(
|
||||||
|
cog_balances("552025209777", 2019L, category = "Fund Balances")
|
||||||
|
)
|
||||||
|
expect_gt(nrow(r), 0L)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('expenditure_concept_direct_suppressed is NA, not FALSE, when categories are collapsed', {
|
||||||
|
# .detect_direct_suppressed() keys on
|
||||||
|
# paste(year, canonical_govid, category, sep = "\r"). In all-categories
|
||||||
|
# mode every row carries the literal "All Categories" value, so an IG-only
|
||||||
|
# row's key collides with any ordinary Direct row for the same
|
||||||
|
# (year, govid) -- has_direct reads TRUE whenever the government has ANY
|
||||||
|
# direct spending at all, candidate is always empty, and the detector can
|
||||||
|
# never fire. Before the fix this silently reported FALSE, an affirmative
|
||||||
|
# claim the code did not actually compute (finding 1). NA is the honest
|
||||||
|
# answer: cog_explain(x, format = "list") is required here, since without
|
||||||
|
# format = "list" it returns the result tibble, not the provenance list.
|
||||||
|
gov <- "552025209777"
|
||||||
|
t <- cog_spending(gov, 2019L, category = "All Categories",
|
||||||
|
expenditure_concept = "total")
|
||||||
|
prov <- cog_explain(t, format = "list")
|
||||||
|
expect_true(is.na(prov$expenditure_concept_direct_suppressed))
|
||||||
|
expect_false(isTRUE(prov$expenditure_concept_direct_suppressed))
|
||||||
|
expect_match(prov$expenditure_concept_note, "unavailable", fixed = TRUE)
|
||||||
|
|
||||||
|
# A per-category "total" query on the same government/year is unaffected --
|
||||||
|
# the detector can still key correctly and reports a strict logical.
|
||||||
|
t_by_cat <- cog_spending(gov, 2019L, expenditure_concept = "total")
|
||||||
|
prov_by_cat <- cog_explain(t_by_cat, format = "list")
|
||||||
|
expect_false(is.na(prov_by_cat$expenditure_concept_direct_suppressed))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('"All Categories" still signposts coverage gaps (finding 6, final whole-branch review)', {
|
||||||
|
# .build_suggestions()'s candidate sub-select used to be keyed on
|
||||||
|
# `category`, e.g. `WHERE category IN ('All Categories')`. Since
|
||||||
|
# .ALL_CATEGORIES is never itself a row in summary_categories.category,
|
||||||
|
# that sub-select always came back empty in all-categories mode, so
|
||||||
|
# `candidates` was empty and .build_suggestions() short-circuited to
|
||||||
|
# list() -- coverage signposting was structurally impossible for the one
|
||||||
|
# mode whose whole selling point is "you cannot sum the wrong scope"
|
||||||
|
# (uscogdata#9's entire point, silently defeated).
|
||||||
|
#
|
||||||
|
# AL state government, FY2011, category = "Corrections": this category has
|
||||||
|
# no legacy leaf rows in FY2011 (aggregate-flagged E04/E05 family), so the
|
||||||
|
# per-category query returns 0 rows and 3 recipe-hint suggestions fire
|
||||||
|
# (empty_year path). All-categories mode does not have an empty year --
|
||||||
|
# the government has other primary spending in FY2011 -- but the same
|
||||||
|
# suppressed Corrections dollars are still excluded from the summed total,
|
||||||
|
# so the fix (scoping the candidate sub-select by subtype_col/subtype_scope
|
||||||
|
# instead of by category, symmetric with .build_verb_sql()) must still
|
||||||
|
# surface them via the suppressed_component path.
|
||||||
|
gov <- "010000226085"
|
||||||
|
|
||||||
|
by_cat <- suppressMessages(cog_spending(gov, 2011L, category = "Corrections"))
|
||||||
|
sugg_by_cat <- cog_explain(by_cat, format = "list")$suggestions
|
||||||
|
expect_gt(length(sugg_by_cat), 0L)
|
||||||
|
|
||||||
|
all_cat <- suppressMessages(cog_spending(gov, 2011L, category = "All Categories"))
|
||||||
|
sugg_all_cat <- cog_explain(all_cat, format = "list")$suggestions
|
||||||
|
expect_gt(length(sugg_all_cat), 0L)
|
||||||
|
|
||||||
|
# The same Corrections recipe that fired per-category must also fire in
|
||||||
|
# all-categories mode -- not just some unrelated recipe.
|
||||||
|
ids_by_cat <- vapply(sugg_by_cat, function(s) s$recipe_id %||% "", character(1))
|
||||||
|
ids_all_cat <- vapply(sugg_all_cat, function(s) s$recipe_id %||% "", character(1))
|
||||||
|
expect_true("corrections_combined" %in% ids_by_cat)
|
||||||
|
expect_true("corrections_combined" %in% ids_all_cat)
|
||||||
|
|
||||||
|
# In all-categories mode the government DOES have other primary spending
|
||||||
|
# in FY2011 (the year itself is not a gap), so the suggestion can only have
|
||||||
|
# fired via the suppressed_component path, not empty_year.
|
||||||
|
corr_all <- sugg_all_cat[[which(ids_all_cat == "corrections_combined")]]
|
||||||
|
expect_identical(corr_all$trigger, "suppressed_component")
|
||||||
|
expect_gt(corr_all$suppressed_amount, 0)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('"All Categories" candidate scoping is symmetric with .build_verb_sql() -- subtype, not category', {
|
||||||
|
# Direct assertion on the mechanism itself (finding 6): in all-categories
|
||||||
|
# mode .build_suggestions() must scope its candidate recipe sub-select by
|
||||||
|
# subtype_col/subtype_scope, not by the literal "All Categories" value.
|
||||||
|
# Passing all_categories = FALSE with the identical category value proves
|
||||||
|
# the branch -- not merely the subtype_col/subtype_scope arguments' mere
|
||||||
|
# presence -- is what changes the query.
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
|
||||||
|
none <- uscogdata:::.build_suggestions(
|
||||||
|
con, cohort = uscogdata:::.make_cohort("010000226085"), years = 2011L,
|
||||||
|
category = "All Categories", result = NULL, basis = "harmonized",
|
||||||
|
flow_prefixes = c("E", "F", "G"),
|
||||||
|
long_view = "spending_long_harmonized",
|
||||||
|
all_categories = FALSE,
|
||||||
|
subtype_col = "spend_subtype",
|
||||||
|
subtype_scope = c("operations", "capital", "assistance")
|
||||||
|
)
|
||||||
|
expect_length(none, 0L)
|
||||||
|
|
||||||
|
scoped <- uscogdata:::.build_suggestions(
|
||||||
|
con, cohort = uscogdata:::.make_cohort("010000226085"), years = 2011L,
|
||||||
|
category = "All Categories", result = NULL, basis = "harmonized",
|
||||||
|
flow_prefixes = c("E", "F", "G"),
|
||||||
|
long_view = "spending_long_harmonized",
|
||||||
|
all_categories = TRUE,
|
||||||
|
subtype_col = "spend_subtype",
|
||||||
|
subtype_scope = c("operations", "capital", "assistance")
|
||||||
|
)
|
||||||
|
expect_gt(length(scoped), 0L)
|
||||||
|
})
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
# Madison walkthrough audit -- finding F-004. Tracked as uscogdata#15.
|
||||||
|
# See docs/walkthroughs/FINDINGS.md in cog_explorer.
|
||||||
|
#
|
||||||
|
# The raw Census files report thousands of dollars; this package multiplies by
|
||||||
|
# 1000 and returns full US dollars. That is the friendlier choice and is not
|
||||||
|
# wrong -- but cog_explorer's CLAUDE.md states "All raw `amt` values are in
|
||||||
|
# $1,000s", so a reader who applies that rule to amt_nominal overstates every
|
||||||
|
# figure by 1000x, and gets a plausible-looking number rather than an obvious
|
||||||
|
# error. The audit rates this the highest-consequence definitional gap it found.
|
||||||
|
#
|
||||||
|
# Deliberately NOT asserted here: man/cog_spending.Rd and man/cog_revenue.Rd,
|
||||||
|
# which ALREADY carry the statement in their @return sections (verified
|
||||||
|
# 2026-07-29), as does cog-api's data-dictionary.md (since 2b71b41). The gap is
|
||||||
|
# in the surfaces a reader meets first and in cog_explorer's own conventions
|
||||||
|
# doc -- see uscogdata#15 for the full surface-by-surface table and for the two
|
||||||
|
# secondary tasks (cog_explorer/CLAUDE.md, which has no git remote, and
|
||||||
|
# cog-api's llms.txt, which is silent on units).
|
||||||
|
|
||||||
|
test_that("returned amounts are documented as full US dollars where readers meet the package", {
|
||||||
|
|
||||||
|
# README and vignettes ship only in the source tree, not in the installed
|
||||||
|
# package, so these assertions cannot run under R CMD check -- CI's earlier
|
||||||
|
# testthat::test_local() step is what enforces them. See
|
||||||
|
# skip_if_no_source_tree() in helper-fixture.R.
|
||||||
|
docs <- skip_if_no_source_tree(
|
||||||
|
"README.md",
|
||||||
|
c("vignettes", "total-spending.Rmd"),
|
||||||
|
c("vignettes", "population-denominators.Rmd")
|
||||||
|
)
|
||||||
|
|
||||||
|
says_units <- function(path) {
|
||||||
|
txt <- paste(readLines(path, warn = FALSE), collapse = " ")
|
||||||
|
grepl("full US dollars|full U\\.S\\. dollars", txt, ignore.case = TRUE) &&
|
||||||
|
grepl("\\$1,000s|thousands of dollars", txt, ignore.case = TRUE)
|
||||||
|
}
|
||||||
|
|
||||||
|
for (path in docs) expect_true(says_units(path))
|
||||||
|
|
||||||
|
# Pin the documented claim to the actual behaviour, so the two cannot drift.
|
||||||
|
# The expected raw amount is read straight from the corpus's parquet
|
||||||
|
# partitions -- never through cog_spending(), which is the thing being
|
||||||
|
# described. Madison FY2020: E/F/G = 623,347 ($1,000s) -> $623,347,000.
|
||||||
|
raw_thousands <- wt_raw_amt("552025209777", 2020L, prefixes = c("E", "F", "G"))
|
||||||
|
expect_equal(raw_thousands, 623347)
|
||||||
|
|
||||||
|
returned <- cog_spending(govid = "552025209777", years = 2020L)
|
||||||
|
expect_equal(sum(returned$amt_nominal), raw_thousands * 1000)
|
||||||
|
|
||||||
|
units <- attr(returned, "provenance")$transformations$units_conversion
|
||||||
|
expect_true(units$applied)
|
||||||
|
expect_equal(units$multiplier, 1000)
|
||||||
|
})
|
||||||
@@ -0,0 +1,539 @@
|
|||||||
|
test_that("balance views register and carry only balance codes", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
con <- cog_open()
|
||||||
|
on.exit(cog_close())
|
||||||
|
|
||||||
|
views <- DBI::dbGetQuery(con,
|
||||||
|
"SELECT table_name FROM information_schema.tables
|
||||||
|
WHERE table_schema = 'main' AND table_type = 'VIEW'"
|
||||||
|
)$table_name
|
||||||
|
expect_true(all(c("balance_long", "balance_annotated") %in% views))
|
||||||
|
|
||||||
|
# Every item_code in balance_long is a category_type = 'balance' member.
|
||||||
|
leak <- DBI::dbGetQuery(con,
|
||||||
|
"SELECT COUNT(*) AS n FROM balance_long
|
||||||
|
WHERE item_code NOT IN (
|
||||||
|
SELECT item_code FROM summary_categories WHERE category_type = 'balance')"
|
||||||
|
)$n
|
||||||
|
expect_identical(as.integer(leak), 0L)
|
||||||
|
|
||||||
|
# balance_annotated exposes the subtype column the verb groups on.
|
||||||
|
cols <- DBI::dbGetQuery(con,
|
||||||
|
"SELECT column_name FROM information_schema.columns
|
||||||
|
WHERE table_name = 'balance_annotated'"
|
||||||
|
)$column_name
|
||||||
|
expect_true(all(c("category", "category_type", "balance_subtype") %in% cols))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("inst/sql/26-balance_long.sql enforces NOT is_aggregate (real SQL text, synthetic parquet)", {
|
||||||
|
# Every category_type = 'balance' item_code in the bundled fixture has
|
||||||
|
# is_aggregate = FALSE for every row of every year -- there is no real row
|
||||||
|
# that would be excluded ONLY by the `AND NOT is_aggregate` predicate. An
|
||||||
|
# assertion against the live fixture (`WHERE is_aggregate` returns 0) is
|
||||||
|
# therefore vacuous: it passes identically whether or not the view's
|
||||||
|
# predicate is present. As with the 22-/23- and 24-/25- tests above, this
|
||||||
|
# reads the real inst/sql/26-balance_long.sql text off disk and executes it
|
||||||
|
# -- plus its 10-long.sql / 11-summary_categories.sql dependencies -- against
|
||||||
|
# a synthetic hive-partitioned parquet tree that DOES contain an aggregate
|
||||||
|
# row under a real balance item_code (W01), so a regression that drops the
|
||||||
|
# predicate changes which rows survive.
|
||||||
|
skip_if_no_corpus()
|
||||||
|
|
||||||
|
tmp <- withr::local_tempdir()
|
||||||
|
part_dir <- file.path(tmp, "data", "long", "year=2004")
|
||||||
|
dir.create(part_dir, recursive = TRUE)
|
||||||
|
part_path <- file.path(part_dir, "part-0.parquet")
|
||||||
|
|
||||||
|
write_con <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
on.exit(DBI::dbDisconnect(write_con, shutdown = TRUE), add = TRUE)
|
||||||
|
DBI::dbExecute(write_con, sprintf("
|
||||||
|
COPY (
|
||||||
|
SELECT * FROM (VALUES
|
||||||
|
('bal-A', 'W01', 100, false), -- control: ordinary balance row, survives
|
||||||
|
('bal-B', 'W01', 999999, true) -- excluded ONLY by `NOT is_aggregate`
|
||||||
|
) AS t(canonical_govid, item_code, amt, is_aggregate)
|
||||||
|
) TO %s (FORMAT PARQUET)
|
||||||
|
", uscogdata:::.sql_lit_chr(part_path)))
|
||||||
|
|
||||||
|
DBI::dbExecute(write_con, sprintf("
|
||||||
|
COPY (
|
||||||
|
SELECT * FROM (VALUES
|
||||||
|
('W01', 'Fund Balances', 'balance', NULL, NULL, 'general')
|
||||||
|
) AS t(item_code, category, category_type, spend_subtype, revenue_subtype, balance_subtype)
|
||||||
|
) TO %s (FORMAT PARQUET)
|
||||||
|
", uscogdata:::.sql_lit_chr(file.path(tmp, "data", "summary_categories.parquet"))))
|
||||||
|
|
||||||
|
sql_dir <- system.file("sql", package = "uscogdata")
|
||||||
|
.read_view_sql <- function(filename) {
|
||||||
|
txt <- paste(readLines(file.path(sql_dir, filename), warn = FALSE), collapse = "\n")
|
||||||
|
uscogdata:::.render_view_sql(txt, paste0(tmp, "/"))
|
||||||
|
}
|
||||||
|
|
||||||
|
con <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE)
|
||||||
|
DBI::dbExecute(con, .read_view_sql("10-long.sql"))
|
||||||
|
DBI::dbExecute(con, .read_view_sql("11-summary_categories.sql"))
|
||||||
|
DBI::dbExecute(con, .read_view_sql("26-balance_long.sql"))
|
||||||
|
|
||||||
|
rows <- DBI::dbGetQuery(con,
|
||||||
|
"SELECT canonical_govid, item_code, amt FROM balance_long ORDER BY canonical_govid"
|
||||||
|
)
|
||||||
|
expect_equal(nrow(rows), 1L)
|
||||||
|
expect_equal(rows$canonical_govid, "bal-A")
|
||||||
|
expect_equal(rows$amt, 100)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("balance views are skipped on a corpus without balance_subtype", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_corpus_missing_balance_subtype({
|
||||||
|
con <- cog_open()
|
||||||
|
on.exit(cog_close())
|
||||||
|
views <- DBI::dbGetQuery(con,
|
||||||
|
"SELECT table_name FROM information_schema.tables
|
||||||
|
WHERE table_schema = 'main' AND table_type = 'VIEW'"
|
||||||
|
)$table_name
|
||||||
|
# Registration must SKIP them, not error -- an older corpus stays usable.
|
||||||
|
expect_false(any(c("balance_long", "balance_annotated") %in% views))
|
||||||
|
expect_true("revenue_long" %in% views)
|
||||||
|
|
||||||
|
# ...and calling the verb on such a corpus must hit
|
||||||
|
# .require_balance_support()'s curated abort (spec § Testing: "Gating"),
|
||||||
|
# not a DuckDB binder error naming a view that was never registered.
|
||||||
|
# Asserted on the CLASS: removing the guard still errors, so a bare
|
||||||
|
# expect_error() would pass on the regression.
|
||||||
|
expect_error(
|
||||||
|
cog_balances("550000227544", 2019),
|
||||||
|
class = "uscogdata_no_balance_support"
|
||||||
|
)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_balances returns holdings for a government that has them", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
r <- cog_balances("550000227544", 2019)
|
||||||
|
expect_s3_class(r, "tbl_df")
|
||||||
|
expect_true(nrow(r) > 0L)
|
||||||
|
expect_true(all(c("year", "canonical_govid", "gov_name", "balance_subtype",
|
||||||
|
"category", "amt_nominal") %in% names(r)))
|
||||||
|
expect_identical(sort(unique(r$category)),
|
||||||
|
c("Fund Balances", "Insurance Trust Balances"))
|
||||||
|
expect_false(is.null(attr(r, "provenance")))
|
||||||
|
expect_identical(attr(r, "provenance")$verb, "cog_balances")
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('category = "Fund Balances" is exactly the general family', {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
r <- cog_balances("550000227544", 2019, category = "Fund Balances")
|
||||||
|
expect_identical(unique(r$balance_subtype), "general")
|
||||||
|
codes <- sort(unlist(strsplit(paste(r$codes_included, collapse = ","), ",")))
|
||||||
|
expect_identical(codes, c("W01", "W31", "W61"))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("no flow code can reach cog_balances", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
r <- cog_balances("550000227544", c(2011, 2012, 2019, 2020))
|
||||||
|
got <- unique(unlist(strsplit(paste(r$codes_included, collapse = ","), ",")))
|
||||||
|
|
||||||
|
# The expected set is read from the RAW corpus, never from the verb --
|
||||||
|
# verifying an absence through the filter that creates it proves nothing.
|
||||||
|
# A fresh, direct DuckDB connection against the raw parquet files (never
|
||||||
|
# cog_open()'s session, never balance_long/balance_annotated) reads
|
||||||
|
# parquet natively -- no arrow dependency needed (see CLAUDE.md).
|
||||||
|
con2 <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
on.exit(DBI::dbDisconnect(con2, shutdown = TRUE), add = TRUE)
|
||||||
|
cats_path <- file.path(fixture_corpus_path(), "data", "summary_categories.parquet")
|
||||||
|
balance_codes <- DBI::dbGetQuery(con2, sprintf(
|
||||||
|
"SELECT item_code FROM read_parquet(%s) WHERE category_type = 'balance'",
|
||||||
|
uscogdata:::.sql_lit_chr(cats_path)
|
||||||
|
))$item_code
|
||||||
|
|
||||||
|
expect_true(length(got) > 0L)
|
||||||
|
expect_true(all(got %in% balance_codes))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("every balance_subtype maps to exactly one category", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
# Dropping the `subtype` argument is only safe while this tree holds. If the
|
||||||
|
# pipeline ever gives a balance subtype a second category, `category` becomes
|
||||||
|
# a lossy filter -- fail HERE rather than in a user's analysis. Read via a
|
||||||
|
# fresh direct DuckDB connection against the raw parquet file, not through
|
||||||
|
# any registered view.
|
||||||
|
con2 <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
on.exit(DBI::dbDisconnect(con2, shutdown = TRUE), add = TRUE)
|
||||||
|
cats_path <- file.path(fixture_corpus_path(), "data", "summary_categories.parquet")
|
||||||
|
b <- DBI::dbGetQuery(con2, sprintf(
|
||||||
|
"SELECT category, balance_subtype FROM read_parquet(%s) WHERE category_type = 'balance'",
|
||||||
|
uscogdata:::.sql_lit_chr(cats_path)
|
||||||
|
))
|
||||||
|
per_subtype <- tapply(b$category, b$balance_subtype,
|
||||||
|
function(x) length(unique(x)))
|
||||||
|
expect_true(all(per_subtype == 1L))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_balances records found + missing govids in provenance", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
suppressMessages(
|
||||||
|
r <- cog_balances(c("550000227544", "XXXINVALID"), 2019)
|
||||||
|
)
|
||||||
|
prov <- attr(r, "provenance")
|
||||||
|
expect_equal(sort(prov$scope$govids_found), "550000227544")
|
||||||
|
expect_equal(sort(prov$scope$govids_missing), "XXXINVALID")
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("per_capita divides holdings by population", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
plain <- cog_balances("550000227544", 2019, category = "Fund Balances")
|
||||||
|
pc <- cog_balances("550000227544", 2019, category = "Fund Balances",
|
||||||
|
per_capita = TRUE)
|
||||||
|
expect_true("amt_per_capita_nominal" %in% names(pc))
|
||||||
|
expect_true("pop_source" %in% names(pc))
|
||||||
|
expect_identical(pc$amt_nominal, plain$amt_nominal)
|
||||||
|
|
||||||
|
# Assert against the denominator read from the corpus, NOT against a
|
||||||
|
# quantity derived from amt_per_capita_nominal itself -- dividing the
|
||||||
|
# column back out would be tautological and would pass on any value.
|
||||||
|
pop <- DBI::dbGetQuery(cog_open(), sprintf(
|
||||||
|
"SELECT population FROM gov_population_yearly
|
||||||
|
WHERE canonical_govid = %s AND year = 2019",
|
||||||
|
uscogdata:::.sql_lit_chr("550000227544")
|
||||||
|
))$population
|
||||||
|
expect_length(pop, 1L)
|
||||||
|
expect_equal(pc$amt_per_capita_nominal, pc$amt_nominal / pop,
|
||||||
|
tolerance = 1e-8)
|
||||||
|
|
||||||
|
prov <- attr(pc, "provenance")
|
||||||
|
expect_true(prov$transformations$per_capita$applied)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("adjust_to_year adds real dollars", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
r <- cog_balances("550000227544", 2012, category = "Fund Balances",
|
||||||
|
adjust_to_year = 2020)
|
||||||
|
expect_true("amt_real" %in% names(r))
|
||||||
|
# 2012 dollars inflated to 2020 must exceed nominal.
|
||||||
|
expect_true(all(r$amt_real > r$amt_nominal))
|
||||||
|
prov <- attr(r, "provenance")
|
||||||
|
expect_true(prov$transformations$inflation$applied)
|
||||||
|
expect_identical(prov$transformations$inflation$base_year, 2020L)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("per_capita and adjust_to_year compose", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
r <- cog_balances("550000227544", 2012, category = "Fund Balances",
|
||||||
|
per_capita = TRUE, adjust_to_year = 2020)
|
||||||
|
expect_true("amt_per_capita_real" %in% names(r))
|
||||||
|
# The per-capita column must be deflated by the SAME factor as the level
|
||||||
|
# column -- this is what the ordering at R/balances.R:101-103 guarantees.
|
||||||
|
# .attach_real_dollars() silently no-ops on the per-capita leg when
|
||||||
|
# amt_per_capita_nominal does not exist yet (R/spending.R:664), so
|
||||||
|
# reversing those two calls drops this column with no error at all.
|
||||||
|
expect_equal(r$amt_per_capita_real / r$amt_per_capita_nominal,
|
||||||
|
r$amt_real / r$amt_nominal, tolerance = 1e-8)
|
||||||
|
|
||||||
|
# And the documented condition is a conjunction: adjust_to_year ALONE
|
||||||
|
# must not produce amt_per_capita_real (pins the @return wording).
|
||||||
|
r2 <- cog_balances("550000227544", 2012, category = "Fund Balances",
|
||||||
|
adjust_to_year = 2020)
|
||||||
|
expect_true("amt_real" %in% names(r2))
|
||||||
|
expect_false("amt_per_capita_real" %in% names(r2))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
# --- input validation ------------------------------------------------------
|
||||||
|
|
||||||
|
test_that("cog_balances validates its inputs like the money verbs", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
G <- "550000227544"
|
||||||
|
# Pinned to the message, not bare expect_error(): every one of these
|
||||||
|
# already produces *some* error or *some* quiet wrong answer today --
|
||||||
|
# years = integer(0) leaks `Parser Error ... AND year IN ()` with the
|
||||||
|
# generated SQL, recipe = c("a","b") throws "the condition has length > 1",
|
||||||
|
# and the govid/category cases return 0 rows with no error at all.
|
||||||
|
expect_error(cog_balances(G, integer(0)), "non-empty integer vector")
|
||||||
|
expect_error(cog_balances(character(0), 2019), "non-empty character vector")
|
||||||
|
expect_error(cog_balances(G, 2019, category = 5), "must be character or NULL")
|
||||||
|
expect_error(cog_balances(G, 2019, recipe = c("a", "b")),
|
||||||
|
"length-1 character string")
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("validation runs after govid coercion, so a data frame still works", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
# .validate_verb_inputs() asserts is.character(govid); it must therefore
|
||||||
|
# run AFTER .coerce_govid_input(), never before, or the documented
|
||||||
|
# data-frame input (cog_gov_search() output) would abort.
|
||||||
|
df <- data.frame(canonical_govid = "550000227544", stringsAsFactors = FALSE)
|
||||||
|
r <- suppressMessages(cog_balances(df, 2019))
|
||||||
|
expect_true(nrow(r) > 0L)
|
||||||
|
expect_identical(unique(r$canonical_govid), "550000227544")
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("recipe and category are mutually exclusive", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
expect_error(
|
||||||
|
cog_balances("550000227544", c(2011, 2012),
|
||||||
|
category = "Fund Balances",
|
||||||
|
recipe = "cash_securities_z77_wide"),
|
||||||
|
class = "uscogdata_recipe_category_conflict"
|
||||||
|
)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
# --- recipe = : the wide-era holdings bridge -------------------------------
|
||||||
|
|
||||||
|
test_that("recipe bridges the wide era into the modern one", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
r <- cog_balances("550000227544", c(2011, 2012),
|
||||||
|
recipe = "cash_securities_z77_wide")
|
||||||
|
# .run_recipe()'s SQL returns `long.year` as a DOUBLE (a corpus-wide trait,
|
||||||
|
# not specific to this recipe -- see the money-verb recipe tests, which
|
||||||
|
# only ever assert on it with expect_equal), so compare numerically rather
|
||||||
|
# than with expect_identical()'s type-strict comparison.
|
||||||
|
expect_equal(sort(r$year), c(2011, 2012))
|
||||||
|
|
||||||
|
# The 2011 leg can ONLY come from X40, which is 100% is_aggregate = TRUE
|
||||||
|
# and therefore invisible to balance_long. If the recipe path ever starts
|
||||||
|
# filtering aggregates, a 45-year series silently truncates to five --
|
||||||
|
# this is the regression guard for phase_r_harmonization_review.md § 0.2.
|
||||||
|
codes <- attr(r, "provenance")$codes_summed$observed
|
||||||
|
expect_true("X40" %in% codes)
|
||||||
|
expect_true("Z77" %in% codes)
|
||||||
|
expect_true(all(r$amt_nominal > 0))
|
||||||
|
|
||||||
|
prov <- attr(r, "provenance")
|
||||||
|
expect_identical(prov$recipe$recipe_id, "cash_securities_z77_wide")
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("the FY2002 book-to-market basis change is disclosed on the recipe path", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
# 2002 is in the year vector deliberately, and must stay -- do not
|
||||||
|
# "simplify" this back to c(2011, 2012).
|
||||||
|
#
|
||||||
|
# .build_series_break_refs() (R/series_breaks.R, shared with every verb)
|
||||||
|
# matches breaks with `break_year BETWEEN min(years) AND max(years)`, and
|
||||||
|
# SB195's break_year is 2002. A c(2011, 2012) span never crosses the
|
||||||
|
# FY2002 book -> market change -- that whole span sits after it, on one
|
||||||
|
# consistent basis -- so NOT disclosing SB195 there is correct behaviour,
|
||||||
|
# not a gap (same reasoning as the "a request that never crosses the
|
||||||
|
# boundary is not affected by it" comment on .build_corpus_break_refs()).
|
||||||
|
#
|
||||||
|
# The property actually worth testing is: a recipe query that observes
|
||||||
|
# X40 AND spans FY2002 discloses SB195. This fixture has no 2002
|
||||||
|
# partition data for X40/Z77 (confirmed: only 2011/2012/2019/2020
|
||||||
|
# partitions exist), so including 2002 in `years` widens the
|
||||||
|
# break-matching window without changing which rows the recipe join
|
||||||
|
# returns -- verified empirically: r$year below is exactly {2011, 2012}
|
||||||
|
# whether or not 2002 is in the request (see task-4-report.md).
|
||||||
|
# Removing 2002 would silently turn this back into the non-crossing case
|
||||||
|
# above and destroy the test's purpose.
|
||||||
|
r <- cog_balances("550000227544", c(2002, 2011, 2012),
|
||||||
|
recipe = "cash_securities_z77_wide")
|
||||||
|
expect_equal(sort(r$year), c(2011, 2012))
|
||||||
|
refs <- attr(r, "provenance")$series_break_refs
|
||||||
|
# SB195 sits on fin_code X40; it can only fire where X40 is observed,
|
||||||
|
# which is exactly the recipe path.
|
||||||
|
expect_true("SB195" %in% refs)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("the second holdings bridge works too", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
# X41 -> Z78, the securities counterpart. Wisconsin carries X41 in 2011
|
||||||
|
# and Z78 in 2012, so both legs are exercised.
|
||||||
|
r <- cog_balances("550000227544", c(2011, 2012),
|
||||||
|
recipe = "cash_securities_z78_wide")
|
||||||
|
codes <- attr(r, "provenance")$codes_summed$observed
|
||||||
|
expect_true(all(c("X41", "Z78") %in% codes))
|
||||||
|
expect_equal(sort(r$year), c(2011, 2012))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("an unknown recipe id is rejected", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
# Asserted on the CLASS .validate_recipe_id() sets (R/recipes.R:84).
|
||||||
|
# Without it the test is non-discriminating: deleting the validation call
|
||||||
|
# leaves .recipe_components() returning 0 rows and comps$label[[1]]
|
||||||
|
# throwing "subscript out of bounds", which a bare expect_error() accepts
|
||||||
|
# while the user loses the curated "valid recipe ids are ..." message.
|
||||||
|
expect_error(
|
||||||
|
cog_balances("550000227544", 2019, recipe = "no_such_recipe"),
|
||||||
|
class = "uscogdata_unknown_recipe"
|
||||||
|
)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
# --- balance_caveats: GAAP disclosure + measured coverage windows ----------
|
||||||
|
|
||||||
|
test_that("balance_caveats is always present and flags the GAAP distinction", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
r <- cog_balances("550000227544", 2019)
|
||||||
|
cav <- attr(r, "provenance")$balance_caveats
|
||||||
|
expect_false(is.null(cav))
|
||||||
|
expect_true(cav$not_gaap)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("coverage_window is computed from the corpus, not hardcoded", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
r <- cog_balances("550000227544", c(2011, 2012, 2019, 2020))
|
||||||
|
cav <- attr(r, "provenance")$balance_caveats
|
||||||
|
|
||||||
|
# Read the "general" family's true year extent independently, via a
|
||||||
|
# fresh DuckDB connection against the raw parquet files (never through
|
||||||
|
# balance_long/.balance_caveats() itself, and never via arrow -- this
|
||||||
|
# package reads parquet through DuckDB only, see CLAUDE.md). Replicates
|
||||||
|
# the same predicates 26-balance_long.sql applies (category_type =
|
||||||
|
# 'balance', NOT is_aggregate) so this is a faithful, independent
|
||||||
|
# measurement rather than a re-statement of the view under test.
|
||||||
|
con2 <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
on.exit(DBI::dbDisconnect(con2, shutdown = TRUE), add = TRUE)
|
||||||
|
long_glob <- file.path(fixture_corpus_path(), "data", "long", "**", "*.parquet")
|
||||||
|
cats_path <- file.path(fixture_corpus_path(), "data", "summary_categories.parquet")
|
||||||
|
obs <- DBI::dbGetQuery(con2, sprintf(
|
||||||
|
"SELECT MIN(l.year) AS y0, MAX(l.year) AS y1
|
||||||
|
FROM read_parquet(%s, hive_partitioning = true) l
|
||||||
|
JOIN read_parquet(%s) c USING (item_code)
|
||||||
|
WHERE c.balance_subtype = 'general' AND NOT l.is_aggregate",
|
||||||
|
uscogdata:::.sql_lit_chr(long_glob), uscogdata:::.sql_lit_chr(cats_path)
|
||||||
|
))
|
||||||
|
|
||||||
|
expect_identical(as.integer(cav$coverage_window$general),
|
||||||
|
c(as.integer(obs$y0), as.integer(obs$y1)))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("coverage_window covers every corpus subtype, not just observed ones", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
# Deliberate contract (provenance-v1.json): the window block is corpus-
|
||||||
|
# scoped so a caller can ask "is there a family I missed?", while
|
||||||
|
# `truncated` is the observed-scoped field. A single-category query must
|
||||||
|
# therefore still report every balance family in the mounted corpus.
|
||||||
|
r <- cog_balances("550000227544", 2019, category = "Fund Balances")
|
||||||
|
expect_identical(unique(r$balance_subtype), "general")
|
||||||
|
|
||||||
|
con2 <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
on.exit(DBI::dbDisconnect(con2, shutdown = TRUE), add = TRUE)
|
||||||
|
cats_path <- file.path(fixture_corpus_path(), "data", "summary_categories.parquet")
|
||||||
|
all_subtypes <- DBI::dbGetQuery(con2, sprintf(
|
||||||
|
"SELECT DISTINCT balance_subtype FROM read_parquet(%s)
|
||||||
|
WHERE balance_subtype IS NOT NULL",
|
||||||
|
uscogdata:::.sql_lit_chr(cats_path)
|
||||||
|
))$balance_subtype
|
||||||
|
|
||||||
|
cav <- attr(r, "provenance")$balance_caveats
|
||||||
|
expect_setequal(names(cav$coverage_window), all_subtypes)
|
||||||
|
expect_true(length(all_subtypes) > 1L)
|
||||||
|
# ...while `truncated` stays scoped to what this query actually observed.
|
||||||
|
expect_true(all(cav$truncated %in% unique(r$balance_subtype)))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("the corpus-constant coverage windows are memoised per session", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
# The windows query has no govid/year predicate: its answer depends only
|
||||||
|
# on which corpus is mounted, so re-running the full balance_long scan on
|
||||||
|
# every call is pure waste (35% of verb runtime on the fixture). Same
|
||||||
|
# memoise-and-invalidate pattern as .uscogdata_env$manifest.
|
||||||
|
expect_null(uscogdata:::.uscogdata_env$balance_coverage_windows)
|
||||||
|
suppressMessages(cog_balances("550000227544", 2019))
|
||||||
|
memo <- uscogdata:::.uscogdata_env$balance_coverage_windows
|
||||||
|
expect_false(is.null(memo))
|
||||||
|
expect_true("general" %in% names(memo))
|
||||||
|
|
||||||
|
uscogdata:::cog_close()
|
||||||
|
expect_null(uscogdata:::.uscogdata_env$balance_coverage_windows)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("a request past a family's coverage window is flagged", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
# employee_retirement (X21/X30/X47/Z77/Z78) genuinely ends at FY2016 in
|
||||||
|
# the LIVE corpus -- Census moved employee retirement reporting to the
|
||||||
|
# Annual Survey of Public Pensions after that year. This bundled FIXTURE
|
||||||
|
# doesn't carry 2013-2016 at all (only 2011/2012/2019/2020 are present),
|
||||||
|
# so the family's *observed* max here is 2012, not 2016. Either way the
|
||||||
|
# requested span (2012, 2019) reaches past what the family covers in
|
||||||
|
# THIS corpus, which is what makes .balance_caveats() flag it -- the
|
||||||
|
# assertion below is about the fixture's measured window, not the FY2016
|
||||||
|
# live-corpus cutoff.
|
||||||
|
r <- cog_balances("550000227544", c(2012, 2019))
|
||||||
|
cav <- attr(r, "provenance")$balance_caveats
|
||||||
|
expect_true("employee_retirement" %in% cav$truncated)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("the provenance schema documents balance_caveats", {
|
||||||
|
sch <- jsonlite::fromJSON(
|
||||||
|
system.file("schemas", "provenance-v1.json", package = "uscogdata"),
|
||||||
|
simplifyVector = FALSE
|
||||||
|
)
|
||||||
|
expect_true("balance_caveats" %in% names(sch$properties))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_explain surfaces the balance caveats", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
# Asserted on the RENDERED text, not on prov$balance_caveats: the field
|
||||||
|
# is already covered above, and the once-per-session cli_inform() means
|
||||||
|
# cog_explain() is the only surface a caller who missed (or suppressed)
|
||||||
|
# the first message can still audit.
|
||||||
|
r <- suppressMessages(cog_balances("550000227544", c(2012, 2019)))
|
||||||
|
# Both streams: cli routes most of its output through conditions that
|
||||||
|
# land on stderr, so a stdout-only capture would be empty (the pattern
|
||||||
|
# used throughout test-explain.R).
|
||||||
|
out <- paste(c(capture.output(cog_explain(r)),
|
||||||
|
capture.output(cog_explain(r), type = "message")),
|
||||||
|
collapse = "\n")
|
||||||
|
expect_match(out, "GAAP")
|
||||||
|
expect_match(out, "employee_retirement")
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_explain on a money-verb result has no balance caveat section", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
r <- suppressMessages(cog_spending("550000227544", 2019))
|
||||||
|
out <- paste(c(capture.output(cog_explain(r)),
|
||||||
|
capture.output(cog_explain(r), type = "message")),
|
||||||
|
collapse = "\n")
|
||||||
|
# Guard against the capture itself being vacuous: the section must be
|
||||||
|
# absent from output that demonstrably contains the rest of the report.
|
||||||
|
expect_match(out, "Data vintage")
|
||||||
|
expect_false(grepl("GAAP", out))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("the caveat message fires once per session", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
expect_message(cog_balances("550000227544", 2019), "not.*GAAP")
|
||||||
|
expect_no_message(cog_balances("550000227544", 2020))
|
||||||
|
})
|
||||||
|
})
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user