Compare commits
22
Commits
303aa59b07
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ba57f9cb27
|
||
|
|
63b421aea2
|
||
|
|
d1165a507e
|
||
|
|
d83301bdbd
|
||
|
|
9adb921170
|
||
|
|
9f5cd988b2
|
||
|
|
9eaa759ccb | ||
|
|
b41d5ee2aa
|
||
|
|
d0d724c4c8
|
||
|
|
56f610ea3f | ||
|
|
62741343ee | ||
|
|
392643bd74
|
||
|
|
0c7c7eb299
|
||
|
|
7274ce3bfe
|
||
|
|
24e86ed598
|
||
|
|
1ec20174b7
|
||
|
|
2e317d3a0f
|
||
|
|
f7c606984f | ||
|
|
9617b86a26
|
||
|
|
224e5d0530 | ||
|
|
5f81ae386b
|
||
|
|
d2caa6de97 |
@@ -3,6 +3,7 @@
|
|||||||
^\.Rproj\.user$
|
^\.Rproj\.user$
|
||||||
^_pkgdown\.yml$
|
^_pkgdown\.yml$
|
||||||
^docs$
|
^docs$
|
||||||
|
^pm$
|
||||||
^Meta$
|
^Meta$
|
||||||
^doc$
|
^doc$
|
||||||
^pkgdown$
|
^pkgdown$
|
||||||
|
|||||||
@@ -39,5 +39,17 @@ jobs:
|
|||||||
echo "PAT_GH is unset -- add it under Settings > Actions > Secrets." >&2
|
echo "PAT_GH is unset -- add it under Settings > Actions > Secrets." >&2
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
|
# On a tag-triggered run, checkout materializes refs/tags/<tag> as a
|
||||||
|
# LIGHTWEIGHT tag at the commit SHA -- the annotated tag object Gitea
|
||||||
|
# holds is never fetched. Mirroring that strips the annotation, and the
|
||||||
|
# NEXT run on main (which does fetch the real object) is then rejected
|
||||||
|
# with "already exists" trying to correct it, because git will not
|
||||||
|
# clobber an existing tag. That is why v0.4.0 failed to mirror.
|
||||||
|
#
|
||||||
|
# Re-fetch canonical tag objects from Gitea first. --force here rewrites
|
||||||
|
# LOCAL tag refs only; it is not a force push and does not weaken the
|
||||||
|
# non-force guarantee on main documented above.
|
||||||
|
git fetch --tags --force origin
|
||||||
|
|
||||||
git push "https://x-access-token:${PAT_GH}@github.com/civilytics/uscogdata.git" \
|
git push "https://x-access-token:${PAT_GH}@github.com/civilytics/uscogdata.git" \
|
||||||
HEAD:refs/heads/main --tags
|
HEAD:refs/heads/main --tags
|
||||||
|
|||||||
@@ -44,6 +44,29 @@ jobs:
|
|||||||
|
|
||||||
- uses: r-lib/actions/setup-pandoc@v2
|
- uses: r-lib/actions/setup-pandoc@v2
|
||||||
|
|
||||||
|
- name: Drop the unused google-chrome apt source
|
||||||
|
if: runner.os == 'Linux'
|
||||||
|
run: |
|
||||||
|
set -x
|
||||||
|
grep -rl 'dl\.google\.com' /etc/apt/sources.list.d/ /etc/apt/sources.list 2>/dev/null || true
|
||||||
|
sudo sh -c "grep -rl 'dl\.google\.com' /etc/apt/sources.list.d/ /etc/apt/sources.list 2>/dev/null | xargs -r rm -f"
|
||||||
|
grep -rl 'dl\.google\.com' /etc/apt/sources.list.d/ /etc/apt/sources.list 2>/dev/null && exit 1 || true
|
||||||
|
# setup-r@v2 runs `sudo apt-get update` before installing R, and that
|
||||||
|
# command fails outright if ANY configured apt source is broken --
|
||||||
|
# even one we never use. The google-chrome source baked into GitHub's
|
||||||
|
# ubuntu-latest image intermittently serves a stale Packages.gz that
|
||||||
|
# doesn't match its own Release file's hash (a Google CDN sync race,
|
||||||
|
# not anything about R or this repo), which took down every ubuntu
|
||||||
|
# leg of this matrix on 2026-09-09. `rm -f` on a guessed filename
|
||||||
|
# (google-chrome.list) reported success but removed nothing -- the
|
||||||
|
# file it actually is on this image apparently doesn't match that
|
||||||
|
# name, since the source kept showing up in setup-r's `apt-get
|
||||||
|
# update` afterward. Find-by-content instead of guessing the
|
||||||
|
# filename, and fail loudly here (before setup-r even runs) if a
|
||||||
|
# matching source is still present, so a future runner-image change
|
||||||
|
# surfaces as a clear failure in this step instead of a confusing
|
||||||
|
# one in setup-r.
|
||||||
|
|
||||||
- uses: r-lib/actions/setup-r@v2
|
- uses: r-lib/actions/setup-r@v2
|
||||||
with:
|
with:
|
||||||
r-version: ${{ matrix.config.r }}
|
r-version: ${{ matrix.config.r }}
|
||||||
|
|||||||
@@ -0,0 +1,64 @@
|
|||||||
|
# Explain the mirror contribution flow on every incoming pull request.
|
||||||
|
#
|
||||||
|
# This repository is a MIRROR. A PR opened here is landed on the canonical Gitea
|
||||||
|
# repository and syncs back; because the merge preserves the contributor's
|
||||||
|
# commits at their original SHAs, GitHub marks the PR "Merged" on its own as
|
||||||
|
# soon as the mirror syncs -- with nobody visibly clicking Merge.
|
||||||
|
#
|
||||||
|
# Without this comment, that reads as a rejection: the contributor sees their PR
|
||||||
|
# close with no review, no merge button pressed, and no explanation. It is
|
||||||
|
# actually the successful outcome. Say so up front, before it happens.
|
||||||
|
#
|
||||||
|
# WHY pull_request_target AND NOT pull_request:
|
||||||
|
# a `pull_request` run from a fork gets a read-only token, so it cannot post a
|
||||||
|
# comment -- which is exactly the case this workflow exists to serve.
|
||||||
|
# `pull_request_target` runs in the context of the BASE repo and gets a writable
|
||||||
|
# token. That is only safe because this job never checks out or executes the
|
||||||
|
# contributor's code; it posts a fixed string. Do not add a checkout of
|
||||||
|
# `github.event.pull_request.head.sha` here -- that combination is the standard
|
||||||
|
# pull_request_target privilege-escalation hole.
|
||||||
|
name: Explain the mirror flow
|
||||||
|
|
||||||
|
on:
|
||||||
|
pull_request_target:
|
||||||
|
types: [opened]
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
pull-requests: write
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
comment:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Post the contribution-flow explainer
|
||||||
|
uses: actions/github-script@v7
|
||||||
|
with:
|
||||||
|
script: |
|
||||||
|
const body = [
|
||||||
|
"Thanks for this — and one thing worth knowing before it happens.",
|
||||||
|
"",
|
||||||
|
"**This repository is a mirror.** Development happens on Gitea at",
|
||||||
|
"`gitea.civilytics.org/Civilytics/uscogdata`. Your pull request will be fetched",
|
||||||
|
"from here, landed there, and synced back.",
|
||||||
|
"",
|
||||||
|
"Because that merge preserves your commits at their original SHAs, **GitHub will",
|
||||||
|
"mark this pull request \"Merged\" on its own** — without anyone visibly clicking",
|
||||||
|
"the Merge button, and possibly without a review comment on this page first.",
|
||||||
|
"",
|
||||||
|
"> If your pull request closes as \"Merged\" and nobody appears to have merged it,",
|
||||||
|
"> that is the normal, successful outcome — not a rejection.",
|
||||||
|
"",
|
||||||
|
"If it is *not* going to be merged, you will get an actual reply saying so.",
|
||||||
|
"",
|
||||||
|
"Substantial contributions get a `ctb` entry in `DESCRIPTION`, which surfaces in",
|
||||||
|
"`citation(\"uscogdata\")`. There is no CLA and no DCO sign-off.",
|
||||||
|
"",
|
||||||
|
"Full details: [CONTRIBUTING.md](https://github.com/civilytics/uscogdata/blob/main/CONTRIBUTING.md).",
|
||||||
|
].join("\n");
|
||||||
|
|
||||||
|
await github.rest.issues.createComment({
|
||||||
|
owner: context.repo.owner,
|
||||||
|
repo: context.repo.repo,
|
||||||
|
issue_number: context.payload.pull_request.number,
|
||||||
|
body,
|
||||||
|
});
|
||||||
@@ -4,6 +4,10 @@
|
|||||||
.Ruserdata
|
.Ruserdata
|
||||||
*.Rproj
|
*.Rproj
|
||||||
inst/doc
|
inst/doc
|
||||||
|
# pkgdown output. Compass used to keep its files in docs/pm/ and
|
||||||
|
# docs/decisions/, which forced this to be written as children with two
|
||||||
|
# re-includes -- git cannot re-include anything beneath an excluded directory.
|
||||||
|
# Compass lives in pm/ now, so the whole directory can be excluded again.
|
||||||
docs/
|
docs/
|
||||||
/doc/
|
/doc/
|
||||||
/Meta/
|
/Meta/
|
||||||
@@ -12,3 +16,6 @@ docs/
|
|||||||
|
|
||||||
# SDD working artifacts (ledger, briefs, review packages) — plans/ stays tracked
|
# SDD working artifacts (ledger, briefs, review packages) — plans/ stays tracked
|
||||||
.superpowers/sdd/
|
.superpowers/sdd/
|
||||||
|
.compass-cache/
|
||||||
|
# roborev snapshots
|
||||||
|
/.roborev/
|
||||||
|
|||||||
@@ -0,0 +1,60 @@
|
|||||||
|
# roborev configuration, initialised by compass.
|
||||||
|
# Reviews are queued to a background daemon -- they never block a commit.
|
||||||
|
|
||||||
|
post_commit_review = 'commit'
|
||||||
|
excluded_commit_patterns = ['WIP', 'chore:', 'chore(', 'docs:', 'Merge ']
|
||||||
|
|
||||||
|
review_guidelines = '''
|
||||||
|
# --- compass:begin (generated -- edit the sources, not this) ---
|
||||||
|
- Prefer returning new values to mutating arguments in place. A function that edits
|
||||||
|
its caller's object is a bug waiting for a second caller.
|
||||||
|
- Validate at system boundaries -- user input, API responses, file contents, config.
|
||||||
|
Fail fast with a message naming the field and the file.
|
||||||
|
- Never swallow an error. Handle it or let it propagate; a bare catch that continues
|
||||||
|
is worse than a crash.
|
||||||
|
- No hardcoded secrets, tokens, or credentials, and no secrets in log output or error
|
||||||
|
messages.
|
||||||
|
- Parameterise every query. String-built SQL is a defect even when the input looks safe.
|
||||||
|
- Keep functions under roughly 50 lines and files under roughly 400. Flag nesting
|
||||||
|
deeper than four levels.
|
||||||
|
- No magic numbers or hardcoded paths -- name them as constants or read them from config.
|
||||||
|
- New behaviour needs a test. A bug fix needs a test that fails without the fix.
|
||||||
|
- Prose a person reads -- an issue title or body, a journal entry, a decision record,
|
||||||
|
the narrative on the status board -- names the action or the thing, not the shape of
|
||||||
|
the machinery. Flag "gate", "seam", "surface area", "load-bearing", "first-class",
|
||||||
|
"primitive", "blast radius". A project's own defined vocabulary is not the target.
|
||||||
|
- Use the native pipe `|>`, not magrittr `%>%`.
|
||||||
|
- snake_case for objects and functions; UPPER_SNAKE for constants. Never use `.` as a
|
||||||
|
word separator in a function name -- it collides with S3 dispatch.
|
||||||
|
- Validate arguments at the top of exported functions with `stopifnot()` or an explicit
|
||||||
|
check, and say which argument was wrong.
|
||||||
|
- Never `setDT()`, `set()`, or otherwise modify by reference a data.table the caller
|
||||||
|
still owns. `as.data.table()` copies; use it.
|
||||||
|
- Prefer `vapply()` to `sapply()` -- `sapply()` silently returns a list when the type
|
||||||
|
varies, which turns a type error into a downstream mystery.
|
||||||
|
- Use `seq_len(n)` / `seq_along(x)`, never `1:n`, which iterates backwards when n is 0.
|
||||||
|
- Compare strings with `==` only after checking for NA; use `identical()` for scalars
|
||||||
|
where NA would be wrong.
|
||||||
|
- Do not call `library()` inside package or module files; attach packages in scripts and
|
||||||
|
test helpers only.
|
||||||
|
- Namespace-qualify calls into other packages (`stats::sd`) in code that is sourced.
|
||||||
|
- Every exported function needs roxygen with `@param` for each argument (type, meaning,
|
||||||
|
and why the default is what it is) and `@return`. Add `@examples` for exported API.
|
||||||
|
- Declare dependencies in DESCRIPTION. Prefer base R or an existing dependency over
|
||||||
|
adding a new one; a package with zero hard deps is worth keeping that way.
|
||||||
|
- Signal errors with `stop()` carrying a condition class, so callers can catch the kind
|
||||||
|
rather than matching on message text.
|
||||||
|
- Keep internals internal. Export only what a user needs; an accidentally exported
|
||||||
|
helper becomes an API you have to keep.
|
||||||
|
- Tests use testthat edition 3. Each test is self-sufficient -- no reliance on state
|
||||||
|
left by an earlier test or on a fixture built elsewhere in the file.
|
||||||
|
- Prefer duplication in tests over a helper that hides what is being asserted.
|
||||||
|
- Every verb calls .ensure_session() first, then queries via DBI::dbGetQuery().
|
||||||
|
- A verb's return value is always a tbl_df carrying a provenance attribute.
|
||||||
|
- govid inputs always go through .coerce_govid_input(); it accepts a character vector or a data frame.
|
||||||
|
- SQL has two layers: view definitions are numbered .sql files in inst/sql/ registered by .register_views(); query construction is inline sprintf() in R. Add a view as a file; build a query in R.
|
||||||
|
- No arrow dependency -- DuckDB reads parquet natively.
|
||||||
|
- withr is Suggests-only and must appear in tests alone.
|
||||||
|
- Tests must pass offline against the bundled fixture; tests/testthat/setup.R sets USCOGDATA_URL for that.
|
||||||
|
# --- compass:end ---
|
||||||
|
'''
|
||||||
@@ -5,11 +5,12 @@
|
|||||||
R package providing a curated reader API for the Civilytics US Census of
|
R package providing a curated reader API for the Civilytics US Census of
|
||||||
Governments finance corpus. Reads Hive-partitioned parquet + `manifest.json`
|
Governments finance corpus. Reads Hive-partitioned parquet + `manifest.json`
|
||||||
published by `cog_pipeline` via DuckDB (local path or remote URL). This is a
|
published by `cog_pipeline` via DuckDB (local path or remote URL). This is a
|
||||||
standalone Gitea repo, sibling to `cog_explorer/cog_pipeline/`.
|
standalone Gitea repo, sibling to the pipeline repo,
|
||||||
|
`Civilytics/census_of_governments_finance_pipeline`, cloned beside this one.
|
||||||
|
|
||||||
**Gitea remote:** `gitea.civilytics.org/Civilytics/uscogdata`
|
**Gitea remote:** `gitea.civilytics.org/Civilytics/uscogdata`
|
||||||
**Full implementation plan:** `../cog_pipeline/docs/plan_phase_n_tasks.md` (Tasks 2.1–2.8 + Phase 3)
|
**Full implementation plan (archived):** `../census_of_governments_finance_pipeline/docs/archive/plan_phase_n_tasks.md` (Tasks 2.1–2.8 + Phase 3)
|
||||||
**Reader contract spec:** `../cog_pipeline/docs/reader-specification.md`
|
**Reader contract spec:** `../census_of_governments_finance_pipeline/docs/reader-specification.md`
|
||||||
|
|
||||||
## Architecture
|
## Architecture
|
||||||
|
|
||||||
|
|||||||
+10
-2
@@ -101,5 +101,13 @@ but "usually" is not a release gate.
|
|||||||
variables set. This is the only check that catches a
|
variables set. This is the only check that catches a
|
||||||
corpus-unreachable defect, and its absence is why 0.3.0 needed fixing.
|
corpus-unreachable defect, and its absence is why 0.3.0 needed fixing.
|
||||||
8. Bump `Version` and add a `NEWS.md` section.
|
8. Bump `Version` and add a `NEWS.md` section.
|
||||||
9. Tag, then update the r-universe registry pin at
|
9. Tag on **Gitea** (`git tag -a vX.Y.Z && git push origin vX.Y.Z`). The mirror
|
||||||
`github.com/civilytics/civilytics.r-universe.dev`.
|
workflow carries tags to GitHub on its own — confirm the tag appears at
|
||||||
|
`github.com/civilytics/uscogdata/tags` before continuing.
|
||||||
|
10. Update the r-universe registry pin at
|
||||||
|
`github.com/civilytics/civilytics.r-universe.dev` — edit `packages.json`'s
|
||||||
|
`branch` to the new tag. **r-universe will not pick up a release until this
|
||||||
|
is edited**: the pin is a tag, deliberately, so a mid-refactor `main` is
|
||||||
|
never published as a release. `"branch": "*release"` would track releases
|
||||||
|
automatically, but it needs a GitHub *Release* object and the mirror pushes
|
||||||
|
tags only — so it would silently never update.
|
||||||
|
|||||||
+76
-6
@@ -84,22 +84,92 @@
|
|||||||
#' `n_units_reporting = 0`, which is precisely the disclosure a silently
|
#' `n_units_reporting = 0`, which is precisely the disclosure a silently
|
||||||
#' missing year fails to make.
|
#' missing year fails to make.
|
||||||
#'
|
#'
|
||||||
#' `n_units_reporting` describes the result the caller actually received, so
|
#' Three counters are returned, each answering a different question:
|
||||||
#' under `coverage = "consistent"` it reports the balanced count. `is_census_year`
|
#'
|
||||||
#' is a statement about the SURVEY CALENDAR, never a claim of completeness:
|
#' * `n_units_expected` -- the universe the caller named (govids passed in,
|
||||||
#' FY1967 is a census year in which only 97 of Wisconsin's 608 cities report.
|
#' or peers for cog_peer_compare). "How many governments did you ask
|
||||||
#' `n_units_reporting` is the number that tells the truth.
|
#' about?"
|
||||||
|
#' * `n_units_collected` -- how many of those appear in the corpus at all
|
||||||
|
#' that year, in ANY category. This is a statement about survey collection,
|
||||||
|
#' independent of what was asked for: "of the governments you named, how
|
||||||
|
#' many did Census actually collect data from this year?" It separates
|
||||||
|
#' sampling (not collected) from real zeros (collected but spends nothing
|
||||||
|
#' in your category).
|
||||||
|
#' * `n_units_reporting` -- how many of those appear with rows for the
|
||||||
|
#' SPECIFIC category you requested. This is always <= n_units_collected:
|
||||||
|
#' a government can be collected but have no rows for "Police" because it
|
||||||
|
#' contracts policing to the county sheriff, not because it wasn't
|
||||||
|
#' surveyed.
|
||||||
|
#'
|
||||||
|
#' `n_units_reporting` therefore conflates two very different things: a unit
|
||||||
|
#' that was not collected (sampling) and a unit that was collected but spends
|
||||||
|
#' nothing in that category. The ratio n_units_collected / n_units_expected is
|
||||||
|
#' the true collection rate; n_units_reporting / n_units_collected measures
|
||||||
|
#' category participation among collected units.
|
||||||
|
#'
|
||||||
|
#' `is_census_year` is a statement about the SURVEY CALENDAR, never a claim of
|
||||||
|
#' completeness: FY1967 is a census year in which only 97 of Wisconsin's 608
|
||||||
|
#' cities report. The counters are what tell the truth.
|
||||||
|
#'
|
||||||
|
#' @param con Active DuckDB connection (used to look up n_units_collected).
|
||||||
|
#' @param long_view The verb's own long view, used for the collection query;
|
||||||
|
#' NULL skips the lookup and leaves n_units_collected as NA_integer_.
|
||||||
|
#' @param expected_ids The full EXPECTED cohort (govids the caller named),
|
||||||
|
#' used as the candidate list for the collection query. Required alongside
|
||||||
|
#' `con`/`long_view` for a correct count -- see the note below on why it
|
||||||
|
#' must not be derived from `result`/`rows`. `NULL`, or non-`NULL` but
|
||||||
|
#' empty after dropping `NA`/`""` entries, skips the lookup and leaves
|
||||||
|
#' n_units_collected as NA_integer_.
|
||||||
#' @noRd
|
#' @noRd
|
||||||
.coverage_table <- function(result, years, n_expected,
|
.coverage_table <- function(result, years, n_expected,
|
||||||
id_col = "canonical_govid", rows = NULL) {
|
id_col = "canonical_govid", rows = NULL,
|
||||||
|
con = NULL, long_view = NULL,
|
||||||
|
expected_ids = NULL) {
|
||||||
years <- sort(unique(as.integer(years)))
|
years <- sort(unique(as.integer(years)))
|
||||||
src <- if (is.null(rows)) result else rows
|
src <- if (is.null(rows)) result else rows
|
||||||
reporting <- vapply(years, function(y) {
|
reporting <- vapply(years, function(y) {
|
||||||
ids <- src[[id_col]][as.integer(src$year) == y]
|
ids <- src[[id_col]][as.integer(src$year) == y]
|
||||||
length(unique(ids[!is.na(ids)]))
|
length(unique(ids[!is.na(ids)]))
|
||||||
}, integer(1))
|
}, integer(1))
|
||||||
|
|
||||||
|
# n_units_collected: count EXPECTED cohort members present in the corpus
|
||||||
|
# for ANY category that year, not just the requested one. This separates
|
||||||
|
# sampling (not collected at all) from real zeros (collected but no rows
|
||||||
|
# for this category). Only computed when a connection, long_view, AND
|
||||||
|
# expected_ids are all provided; otherwise NA_integer_.
|
||||||
|
#
|
||||||
|
# The candidate list MUST be expected_ids, not derived from `result`/
|
||||||
|
# `rows`: a government with zero rows in the requested category across
|
||||||
|
# EVERY requested year never appears in `result` at all, so deriving
|
||||||
|
# candidates from it would silently exclude exactly the "collected but
|
||||||
|
# real zero" governments this counter exists to count -- collapsing
|
||||||
|
# n_units_collected back to n_units_reporting for precisely the case #36
|
||||||
|
# was filed over.
|
||||||
|
if (!is.null(con) && !is.null(long_view) && length(expected_ids) > 0L) {
|
||||||
|
cohort_chr <- .sql_lit_chr(unique(expected_ids[!is.na(expected_ids) &
|
||||||
|
nzchar(expected_ids)]))
|
||||||
|
years_lit <- paste(years, collapse = ",")
|
||||||
|
collected_q <- sprintf(
|
||||||
|
"SELECT year, COUNT(DISTINCT canonical_govid) AS n
|
||||||
|
FROM %s
|
||||||
|
WHERE canonical_govid IN (%s)
|
||||||
|
AND year IN (%s)
|
||||||
|
GROUP BY year",
|
||||||
|
long_view, cohort_chr, years_lit
|
||||||
|
)
|
||||||
|
collected_df <- DBI::dbGetQuery(con, collected_q)
|
||||||
|
collected_map <- setNames(collected_df$n, as.integer(collected_df$year))
|
||||||
|
collected <- vapply(years, function(y) {
|
||||||
|
val <- collected_map[as.character(y)]
|
||||||
|
if (is.na(val)) 0L else as.integer(val)
|
||||||
|
}, integer(1))
|
||||||
|
} else {
|
||||||
|
collected <- rep(NA_integer_, length(years))
|
||||||
|
}
|
||||||
|
|
||||||
tibble::tibble(
|
tibble::tibble(
|
||||||
year = years,
|
year = years,
|
||||||
|
n_units_collected = collected,
|
||||||
n_units_reporting = as.integer(reporting),
|
n_units_reporting = as.integer(reporting),
|
||||||
n_units_expected = rep(as.integer(n_expected), length(years)),
|
n_units_expected = rep(as.integer(n_expected), length(years)),
|
||||||
is_census_year = .is_census_year(years)
|
is_census_year = .is_census_year(years)
|
||||||
|
|||||||
+24
-6
@@ -166,12 +166,30 @@ cog_explain <- function(result, format = c("print", "list")) {
|
|||||||
cli::cli_h2("Reporting coverage")
|
cli::cli_h2("Reporting coverage")
|
||||||
cli::cli_text("Mode: {prov$coverage_mode %||% 'all'}")
|
cli::cli_text("Mode: {prov$coverage_mode %||% 'all'}")
|
||||||
cov <- prov$coverage
|
cov <- prov$coverage
|
||||||
cli::cli_ul(sprintf(
|
has_collected <- "n_units_collected" %in% names(cov)
|
||||||
"%d: %d of %d units reporting (%.0f%%) -- %s year",
|
if (has_collected) {
|
||||||
cov$year, cov$n_units_reporting, cov$n_units_expected,
|
# Three counters: collected separates sampling from real zeros;
|
||||||
100 * cov$n_units_reporting / pmax(cov$n_units_expected, 1L),
|
# reporting is category-conditional and never a response rate.
|
||||||
ifelse(cov$is_census_year, "census", "sample")
|
cli::cli_ul(sprintf(
|
||||||
))
|
"%d: %d of %d units collected, %d reporting in this category -- %s year",
|
||||||
|
cov$year,
|
||||||
|
cov$n_units_collected,
|
||||||
|
cov$n_units_expected,
|
||||||
|
cov$n_units_reporting,
|
||||||
|
ifelse(cov$is_census_year, "census", "sample")
|
||||||
|
))
|
||||||
|
} else {
|
||||||
|
cli::cli_ul(sprintf(
|
||||||
|
"%d: %d of %d units reporting (%.0f%%) -- %s year",
|
||||||
|
cov$year, cov$n_units_reporting, cov$n_units_expected,
|
||||||
|
100 * cov$n_units_reporting / pmax(cov$n_units_expected, 1L),
|
||||||
|
ifelse(cov$is_census_year, "census", "sample")
|
||||||
|
))
|
||||||
|
}
|
||||||
|
# Explains what the per-row "-- sample year" tag means, regardless of
|
||||||
|
# which branch above rendered it -- not gated on has_collected, which
|
||||||
|
# would make this permanently unreachable now that both real callers
|
||||||
|
# (cog_geographic_rollup(), cog_peer_compare()) always supply it.
|
||||||
if (any(!cov$is_census_year)) {
|
if (any(!cov$is_census_year)) {
|
||||||
cli::cli_text(
|
cli::cli_text(
|
||||||
"Note: the Census of Governments is a complete census only in years ending in 2 or 7; every other year is a sample."
|
"Note: the Census of Governments is a complete census only in years ending in 2 or 7; every other year is a sample."
|
||||||
|
|||||||
@@ -195,11 +195,13 @@ cog_find_peers <- function(target_govid,
|
|||||||
#' a balanced panel.
|
#' a balanced panel.
|
||||||
#'
|
#'
|
||||||
#' Regardless of mode, `provenance$coverage` always carries per-year
|
#' Regardless of mode, `provenance$coverage` always carries per-year
|
||||||
#' `n_units_reporting`, `n_units_expected` and `is_census_year`, and
|
#' `n_units_expected`, `n_units_collected`, `n_units_reporting` and
|
||||||
#' `provenance$coverage_mode` records the mode. `is_census_year` is a
|
#' `is_census_year`, and `provenance$coverage_mode` records the mode.
|
||||||
#' statement about the **survey calendar**, never a claim of completeness:
|
#' `is_census_year` is a statement about the **survey calendar**, never a
|
||||||
#' FY1967 is a census year in which only 97 of Wisconsin's 608 cities
|
#' claim of completeness: FY1967 is a census year in which only 97 of
|
||||||
#' report. `n_units_reporting` is the number that tells the truth.
|
#' Wisconsin's 608 cities report. `n_units_reporting` is
|
||||||
|
#' category-conditional and is not a response rate on its own -- see
|
||||||
|
#' "Reading `coverage`" below for what each counter answers.
|
||||||
#'
|
#'
|
||||||
#' The comparison target is exempt from `"consistent"` balancing -- it is the
|
#' The comparison target is exempt from `"consistent"` balancing -- it is the
|
||||||
#' subject of the comparison, not a member of the cohort -- and the
|
#' subject of the comparison, not a member of the cohort -- and the
|
||||||
@@ -241,14 +243,23 @@ cog_find_peers <- function(target_govid,
|
|||||||
#' summarise(p50 = quantile(total, 0.5, na.rm = TRUE))
|
#' summarise(p50 = quantile(total, 0.5, na.rm = TRUE))
|
||||||
#' ```
|
#' ```
|
||||||
#' @section Reading `coverage`:
|
#' @section Reading `coverage`:
|
||||||
#' `provenance$coverage` reports `n_units_reporting` against
|
#' `provenance$coverage` carries three per-year counters:
|
||||||
#' `n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
#'
|
||||||
#' it counts cohort members with rows for the category you asked for, not
|
#' * `n_units_expected` -- how many governments you asked about.
|
||||||
#' cohort members collected that year.** A government that was surveyed and
|
#' * `n_units_collected` -- how many of those appear in the corpus at all
|
||||||
#' genuinely spends nothing in that category is indistinguishable here from one
|
#' that year (in ANY category), separating sampling from real zeros.
|
||||||
#' that was never surveyed.
|
#' * `n_units_reporting` -- how many have rows for the SPECIFIC category you
|
||||||
|
#' requested. This is always <= n_units_collected: a government can be
|
||||||
|
#' collected but have no rows for "Police" because it contracts policing
|
||||||
|
#' to the county sheriff, not because it wasn't surveyed.
|
||||||
|
#'
|
||||||
|
#' **`n_units_reporting` is category-conditional** and therefore **not a
|
||||||
|
#' response rate**: `n_units_reporting / n_units_expected` conflates sampling
|
||||||
|
#' (never collected) with real zeros (collected but spends nothing in your
|
||||||
|
#' category). Use `n_units_collected / n_units_expected` for the true
|
||||||
|
#' collection rate, and `n_units_reporting / n_units_collected` for category
|
||||||
|
#' participation among collected units.
|
||||||
#'
|
#'
|
||||||
#' The ratio is therefore **not a response rate** and must not be used as one.
|
|
||||||
#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||||
#' `category = "Police"`; the 174-city gap is overwhelmingly cities that
|
#' `category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||||
#' contract policing to the county sheriff, not non-response.
|
#' contract policing to the county sheriff, not non-response.
|
||||||
@@ -325,10 +336,20 @@ cog_peer_compare <- function(target_govid, peers, category, years,
|
|||||||
# Counted over PEER rows only, against the cohort size: "3 of your 15 peers
|
# Counted over PEER rows only, against the cohort size: "3 of your 15 peers
|
||||||
# reported in FY2019". Including the target would inflate every count by one
|
# reported in FY2019". Including the target would inflate every count by one
|
||||||
# and make a cohort that has entirely stopped reporting look non-empty.
|
# and make a cohort that has entirely stopped reporting look non-empty.
|
||||||
|
# n_units_collected is looked up against the spending long view matching
|
||||||
|
# whatever basis cog_spending() actually resolved above (prov$basis) --
|
||||||
|
# NOT hardcoded to spending_long_harmonized, which does not exist on a
|
||||||
|
# corpus with schema_version < 5 (R/basis.R resolves basis = "raw" there,
|
||||||
|
# and only *_long, not *_long_harmonized, is registered; see R/views.R).
|
||||||
|
# The con comes from .ensure_session() already called inside cog_spending().
|
||||||
|
con <- .ensure_session()
|
||||||
prov$coverage_mode <- coverage
|
prov$coverage_mode <- coverage
|
||||||
prov$coverage <- .coverage_table(
|
prov$coverage <- .coverage_table(
|
||||||
out, years, length(peer_govids),
|
out, years, length(peer_govids),
|
||||||
rows = r[r$role == "peer", , drop = FALSE]
|
rows = r[r$role == "peer", , drop = FALSE],
|
||||||
|
con = con,
|
||||||
|
long_view = .select_long_view("spending_annotated", prov$basis),
|
||||||
|
expected_ids = peer_govids
|
||||||
)
|
)
|
||||||
attr(out, "provenance") <- prov
|
attr(out, "provenance") <- prov
|
||||||
out
|
out
|
||||||
|
|||||||
+36
-13
@@ -49,11 +49,13 @@
|
|||||||
#' a balanced panel.
|
#' a balanced panel.
|
||||||
#'
|
#'
|
||||||
#' Regardless of mode, `provenance$coverage` always carries per-year
|
#' Regardless of mode, `provenance$coverage` always carries per-year
|
||||||
#' `n_units_reporting`, `n_units_expected` and `is_census_year`, and
|
#' `n_units_expected`, `n_units_collected`, `n_units_reporting` and
|
||||||
#' `provenance$coverage_mode` records the mode. `is_census_year` is a
|
#' `is_census_year`, and `provenance$coverage_mode` records the mode.
|
||||||
#' statement about the **survey calendar**, never a claim of completeness:
|
#' `is_census_year` is a statement about the **survey calendar**, never a
|
||||||
#' FY1967 is a census year in which only 97 of Wisconsin's 608 cities
|
#' claim of completeness: FY1967 is a census year in which only 97 of
|
||||||
#' report. `n_units_reporting` is the number that tells the truth.
|
#' Wisconsin's 608 cities report. `n_units_reporting` is
|
||||||
|
#' category-conditional and is not a response rate on its own -- see
|
||||||
|
#' "Reading `coverage`" below for what each counter answers.
|
||||||
#' @return Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`,
|
#' @return Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`,
|
||||||
#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real` /
|
#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real` /
|
||||||
#' `amt_per_capita_nominal` / `amt_per_capita_real`, optional `pop_source`,
|
#' `amt_per_capita_nominal` / `amt_per_capita_real`, optional `pop_source`,
|
||||||
@@ -61,14 +63,23 @@
|
|||||||
#' `provenance` attribute with `verb = "cog_geographic_rollup"`, `layers`,
|
#' `provenance` attribute with `verb = "cog_geographic_rollup"`, `layers`,
|
||||||
#' and `rollup$included_govids` / `rollup$excluded_govids`.
|
#' and `rollup$included_govids` / `rollup$excluded_govids`.
|
||||||
#' @section Reading `coverage`:
|
#' @section Reading `coverage`:
|
||||||
#' `provenance$coverage` reports `n_units_reporting` against
|
#' `provenance$coverage` carries three per-year counters:
|
||||||
#' `n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
#'
|
||||||
#' it counts governments with rows for the category you asked for, not
|
#' * `n_units_expected` -- how many governments you asked about.
|
||||||
#' governments collected that year.** A government that was surveyed and
|
#' * `n_units_collected` -- how many of those appear in the corpus at all
|
||||||
#' genuinely spends nothing in that category is indistinguishable here from one
|
#' that year (in ANY category), separating sampling from real zeros.
|
||||||
#' that was never surveyed.
|
#' * `n_units_reporting` -- how many have rows for the SPECIFIC category you
|
||||||
|
#' requested. This is always <= n_units_collected: a government can be
|
||||||
|
#' collected but have no rows for "Police" because it contracts policing
|
||||||
|
#' to the county sheriff, not because it wasn't surveyed.
|
||||||
|
#'
|
||||||
|
#' **`n_units_reporting` is category-conditional** and therefore **not a
|
||||||
|
#' response rate**: `n_units_reporting / n_units_expected` conflates sampling
|
||||||
|
#' (never collected) with real zeros (collected but spends nothing in your
|
||||||
|
#' category). Use `n_units_collected / n_units_expected` for the true
|
||||||
|
#' collection rate, and `n_units_reporting / n_units_collected` for category
|
||||||
|
#' participation among collected units.
|
||||||
#'
|
#'
|
||||||
#' The ratio is therefore **not a response rate** and must not be used as one.
|
|
||||||
#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||||
#' `category = "Police"`; the 174-city gap is overwhelmingly cities that
|
#' `category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||||
#' contract policing to the county sheriff, not non-response.
|
#' contract policing to the county sheriff, not non-response.
|
||||||
@@ -137,8 +148,20 @@ cog_geographic_rollup <- function(govids, category, years,
|
|||||||
# n_units_expected is the universe the CALLER named -- the govids passed in
|
# n_units_expected is the universe the CALLER named -- the govids passed in
|
||||||
# -- not the national universe. That is what makes the ratio meaningful:
|
# -- not the national universe. That is what makes the ratio meaningful:
|
||||||
# "597 of the 608 Wisconsin cities you asked about reported in FY2012".
|
# "597 of the 608 Wisconsin cities you asked about reported in FY2012".
|
||||||
|
# n_units_collected is looked up against the spending long view matching
|
||||||
|
# whatever basis cog_spending() actually resolved above (prov$basis) --
|
||||||
|
# NOT hardcoded to spending_long_harmonized, which does not exist on a
|
||||||
|
# corpus with schema_version < 5 (R/basis.R resolves basis = "raw" there,
|
||||||
|
# and only *_long, not *_long_harmonized, is registered; see R/views.R).
|
||||||
|
# The con comes from .ensure_session() already called inside cog_spending().
|
||||||
|
con <- .ensure_session()
|
||||||
prov$coverage_mode <- coverage
|
prov$coverage_mode <- coverage
|
||||||
prov$coverage <- .coverage_table(r, years, length(unique(all_govids)))
|
prov$coverage <- .coverage_table(
|
||||||
|
r, years, length(unique(all_govids)),
|
||||||
|
con = con,
|
||||||
|
long_view = .select_long_view("spending_annotated", prov$basis),
|
||||||
|
expected_ids = all_govids
|
||||||
|
)
|
||||||
attr(r, "provenance") <- prov
|
attr(r, "provenance") <- prov
|
||||||
|
|
||||||
r
|
r
|
||||||
|
|||||||
+167
-49
@@ -84,6 +84,21 @@
|
|||||||
#' @return List of `list(recipe_id, label, available_years, hint,
|
#' @return List of `list(recipe_id, label, available_years, hint,
|
||||||
#' ig_recipe_id, trigger, suppressed_amount, suppressed_years,
|
#' ig_recipe_id, trigger, suppressed_amount, suppressed_years,
|
||||||
#' suppressed_codes)`, possibly empty.
|
#' suppressed_codes)`, possibly empty.
|
||||||
|
#'
|
||||||
|
#' Decomposed (Issue #33) into three extracted helpers to stay within the
|
||||||
|
#' project's "functions under 50 lines" convention:
|
||||||
|
#' \itemize{
|
||||||
|
#' \item `.query_candidate_recipes()` -- candidate recipe lookup by
|
||||||
|
#' category/subtype scope + `category_type` filter (#34) + M/L exclusion.
|
||||||
|
#' \item `.query_recipe_meta()` -- metadata (label, year spans).
|
||||||
|
#' \item `.query_covered_years()` -- Path 1 gap-year coverage via the
|
||||||
|
#' recipe's own generic join.
|
||||||
|
#' }
|
||||||
|
#' The for-loop that merges covered-years + suppressed-components into
|
||||||
|
#' suggestion objects stays inline here because it interleaves
|
||||||
|
#' empty_hit/supp_hit precedence with field assembly. Likewise kept inline:
|
||||||
|
#' the M/L-exclusion design-comment block and the final
|
||||||
|
#' `.attach_ig_counterparts()` call.
|
||||||
#' @noRd
|
#' @noRd
|
||||||
.build_suggestions <- function(con, cohort, years, category, result, basis,
|
.build_suggestions <- function(con, cohort, years, category, result, basis,
|
||||||
flow_prefixes, long_view,
|
flow_prefixes, long_view,
|
||||||
@@ -111,28 +126,16 @@
|
|||||||
# by `category` (`.ALL_CATEGORIES` is never a row in
|
# by `category` (`.ALL_CATEGORIES` is never a row in
|
||||||
# `summary_categories.category`, so a category-keyed sub-select always
|
# `summary_categories.category`, so a category-keyed sub-select always
|
||||||
# came back empty here). The M/L exclusion below is unchanged either way.
|
# came back empty here). The M/L exclusion below is unchanged either way.
|
||||||
candidate_scope_sql <- if (isTRUE(all_categories)) {
|
#
|
||||||
sprintf(
|
# Issue #34: scope the candidate query by `category_type` ('expenditure'
|
||||||
"SELECT DISTINCT item_code FROM summary_categories WHERE %s IN (%s)",
|
# vs 'revenue') to prevent cross-flow-family leakage -- e.g.
|
||||||
subtype_col, .sql_lit_chr(subtype_scope)
|
# `cog_revenue(category = "Corrections")` must not surface
|
||||||
)
|
# expenditure-only recipes (E04/E05) merely because they share the same
|
||||||
} else {
|
# category name in summary_categories. The type is derived from
|
||||||
sprintf(
|
# flow_prefixes: E/F/G -> 'expenditure', anything else -> 'revenue'.
|
||||||
"SELECT DISTINCT item_code FROM summary_categories WHERE category IN (%s)",
|
candidates <- .query_candidate_recipes(con, category, flow_prefixes,
|
||||||
.sql_lit_chr(category)
|
all_categories, subtype_col,
|
||||||
)
|
subtype_scope)
|
||||||
}
|
|
||||||
candidates <- DBI::dbGetQuery(con, sprintf(
|
|
||||||
"SELECT DISTINCT recipe_id FROM harmonization_recipes
|
|
||||||
WHERE component_code IN (
|
|
||||||
%s
|
|
||||||
)
|
|
||||||
AND recipe_id NOT IN (
|
|
||||||
SELECT DISTINCT recipe_id FROM harmonization_recipes
|
|
||||||
WHERE LEFT(component_code, 1) IN ('M', 'L')
|
|
||||||
)",
|
|
||||||
candidate_scope_sql
|
|
||||||
))$recipe_id
|
|
||||||
if (length(candidates) == 0L) return(list())
|
if (length(candidates) == 0L) return(list())
|
||||||
|
|
||||||
result_years <- if (is.null(result) || nrow(result) == 0L) {
|
result_years <- if (is.null(result) || nrow(result) == 0L) {
|
||||||
@@ -164,36 +167,11 @@
|
|||||||
|
|
||||||
if (length(gap_years) == 0L && nrow(supp) == 0L) return(list())
|
if (length(gap_years) == 0L && nrow(supp) == 0L) return(list())
|
||||||
|
|
||||||
meta <- tibble::as_tibble(DBI::dbGetQuery(con, sprintf(
|
meta <- .query_recipe_meta(con, candidates)
|
||||||
"SELECT recipe_id, any_value(label) AS label,
|
|
||||||
MIN(year_min) AS year_min, MAX(year_max) AS year_max
|
|
||||||
FROM harmonization_recipes
|
|
||||||
WHERE recipe_id IN (%s)
|
|
||||||
GROUP BY recipe_id",
|
|
||||||
.sql_lit_chr(candidates)
|
|
||||||
)))
|
|
||||||
|
|
||||||
# Path 1 (unchanged): (recipe, year) pairs the recipe's own generic join
|
# Path 1 (unchanged): (recipe, year) pairs the recipe's own generic join
|
||||||
# covers for this government, restricted to the gap years.
|
# covers for this government, restricted to the gap years.
|
||||||
covered <- if (length(gap_years) == 0L) {
|
covered <- .query_covered_years(con, candidates, cohort, gap_years)
|
||||||
data.frame(recipe_id = character(0), year = integer(0))
|
|
||||||
} else {
|
|
||||||
DBI::dbGetQuery(con, sprintf(
|
|
||||||
"SELECT DISTINCT r.recipe_id, l.year
|
|
||||||
FROM long l
|
|
||||||
JOIN harmonization_recipes r
|
|
||||||
ON l.item_code = r.component_code
|
|
||||||
AND l.year BETWEEN r.year_min AND r.year_max
|
|
||||||
AND (r.gov_type_scope = 'all'
|
|
||||||
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
|
||||||
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
|
||||||
WHERE r.recipe_id IN (%s)
|
|
||||||
AND %s
|
|
||||||
AND l.year IN (%s)",
|
|
||||||
.sql_lit_chr(candidates), .cohort_sql(cohort, "l.canonical_govid"),
|
|
||||||
paste(gap_years, collapse = ",")
|
|
||||||
))
|
|
||||||
}
|
|
||||||
|
|
||||||
suggestions <- list()
|
suggestions <- list()
|
||||||
for (rid in candidates) {
|
for (rid in candidates) {
|
||||||
@@ -227,6 +205,146 @@
|
|||||||
.attach_ig_counterparts(con, suggestions, flow_prefixes)
|
.attach_ig_counterparts(con, suggestions, flow_prefixes)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#' Query candidate harmonization recipe IDs for a coverage-gap suggestion.
|
||||||
|
#'
|
||||||
|
#' Selects recipes whose component codes fall within the requested scope
|
||||||
|
#' (category or subtype allowlist), excluding any recipe that is ITSELF an
|
||||||
|
#' intergovernmental (M/L) recipe -- i.e. every one of its own component
|
||||||
|
#' codes is M/L-prefixed. Without this exclusion, a category whose
|
||||||
|
#' summary_categories rows span both a Direct family (e.g. E04/E05,
|
||||||
|
#' "Corrections") and its M/L counterpart (M04/M05) makes the M/L recipe
|
||||||
|
#' itself a raw top-level candidate for a plain `cog_spending()` call --
|
||||||
|
#' following that hint would silently return intergovernmental dollars
|
||||||
|
#' under `expenditure_concept = "direct"` provenance.
|
||||||
|
#'
|
||||||
|
#' In all-categories mode (`all_categories = TRUE`) the inner sub-select is
|
||||||
|
#' scoped by `subtype_col`/`subtype_scope` -- the same allowlist
|
||||||
|
#' `.build_verb_sql()` applies as a WHERE predicate to make the summed
|
||||||
|
#' result a *concept* (see R/spending.R), not by `category`.
|
||||||
|
#' `.ALL_CATEGORIES` ("All Categories") is never itself a row in
|
||||||
|
#' `summary_categories.category`, so a category-keyed sub-select always
|
||||||
|
#' returns zero candidates and silently disables signposting.
|
||||||
|
#'
|
||||||
|
#' Scope is also by `category_type` ('expenditure' vs 'revenue', Issue #34)
|
||||||
|
#' to prevent cross-flow-family leakage: `cog_revenue(category =
|
||||||
|
#' "Corrections")` must not surface expenditure-only recipes (E04/E05)
|
||||||
|
#' merely because they share the same category name in summary_categories.
|
||||||
|
#' The type is derived from flow_prefixes: E/F/G -> 'expenditure', anything
|
||||||
|
#' else -> 'revenue'.
|
||||||
|
#'
|
||||||
|
#' @param con Active DuckDB connection.
|
||||||
|
#' @param category Category name, or `NULL`.
|
||||||
|
#' @param flow_prefixes The calling verb's own flow-type prefixes (see
|
||||||
|
#' `.build_suggestions()`). Used to derive `category_type` (#34).
|
||||||
|
#' @param all_categories `TRUE` when the caller used `.ALL_CATEGORIES`.
|
||||||
|
#' @param subtype_col Name of the summary_categories subtype column to
|
||||||
|
#' scope by when `all_categories = TRUE`; ignored otherwise.
|
||||||
|
#' @param subtype_scope Character vector of subtype values to scope by
|
||||||
|
#' when `all_categories = TRUE`; ignored otherwise.
|
||||||
|
#' @return Character vector of recipe IDs (possibly empty).
|
||||||
|
#' @noRd
|
||||||
|
.query_candidate_recipes <- function(con, category, flow_prefixes,
|
||||||
|
all_categories = FALSE,
|
||||||
|
subtype_col = NULL,
|
||||||
|
subtype_scope = NULL) {
|
||||||
|
# Issue #34: derive category_type from flow_prefixes to prevent
|
||||||
|
# cross-flow-family leakage -- e.g. cog_revenue(category = "Corrections")
|
||||||
|
# must not surface expenditure-only recipes merely because they share the
|
||||||
|
# same category name in summary_categories.
|
||||||
|
category_type <- if (all(flow_prefixes %in% c("E", "F", "G"))) {
|
||||||
|
"expenditure"
|
||||||
|
} else {
|
||||||
|
"revenue"
|
||||||
|
}
|
||||||
|
|
||||||
|
candidate_scope_sql <- if (isTRUE(all_categories)) {
|
||||||
|
sprintf(
|
||||||
|
"SELECT DISTINCT item_code FROM summary_categories
|
||||||
|
WHERE %s IN (%s) AND category_type = '%s'",
|
||||||
|
subtype_col, .sql_lit_chr(subtype_scope), category_type
|
||||||
|
)
|
||||||
|
} else {
|
||||||
|
sprintf(
|
||||||
|
"SELECT DISTINCT item_code FROM summary_categories
|
||||||
|
WHERE category IN (%s) AND category_type = '%s'",
|
||||||
|
.sql_lit_chr(category), category_type
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
DBI::dbGetQuery(con, sprintf(
|
||||||
|
"SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||||
|
WHERE component_code IN (
|
||||||
|
%s
|
||||||
|
)
|
||||||
|
AND recipe_id NOT IN (
|
||||||
|
SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||||
|
WHERE LEFT(component_code, 1) IN ('M', 'L')
|
||||||
|
)",
|
||||||
|
candidate_scope_sql
|
||||||
|
))$recipe_id
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Query gap-year coverage: which (recipe, year) pairs the recipe's own
|
||||||
|
#' generic join covers for this government, restricted to `gap_years`.
|
||||||
|
#'
|
||||||
|
#' This is Path 1 of a suggestion (unchanged): it finds recipes whose
|
||||||
|
#' component codes' generic join produces at least one row for this
|
||||||
|
#' government in each gap year -- i.e. the category returned nothing in
|
||||||
|
#' that year but a recipe would fill it.
|
||||||
|
#'
|
||||||
|
#' @param con Active DuckDB connection.
|
||||||
|
#' @param candidates Character vector of recipe IDs to check coverage for.
|
||||||
|
#' @param cohort The verb's cohort object (see `.make_cohort()`), rendered
|
||||||
|
#' into the govid predicate on the joined `long` scan via `.cohort_sql()`.
|
||||||
|
#' @param gap_years Integer vector of requested years absent from the
|
||||||
|
#' result.
|
||||||
|
#' @return Data frame with columns `recipe_id` (character) and `year`
|
||||||
|
#' (integer). Returns an empty data frame (`recipe_id = character(0)`,
|
||||||
|
#' `year = integer(0)`) when `gap_years` or `candidates` is empty, so
|
||||||
|
#' callers can safely reference `$recipe_id`.
|
||||||
|
#' @noRd
|
||||||
|
.query_covered_years <- function(con, candidates, cohort, gap_years) {
|
||||||
|
if (length(gap_years) == 0L || length(candidates) == 0L) {
|
||||||
|
return(data.frame(recipe_id = character(0), year = integer(0)))
|
||||||
|
}
|
||||||
|
res <- DBI::dbGetQuery(con, sprintf(
|
||||||
|
"SELECT DISTINCT r.recipe_id, l.year
|
||||||
|
FROM long l
|
||||||
|
JOIN harmonization_recipes r
|
||||||
|
ON l.item_code = r.component_code
|
||||||
|
AND l.year BETWEEN r.year_min AND r.year_max
|
||||||
|
AND (r.gov_type_scope = 'all'
|
||||||
|
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||||
|
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||||
|
WHERE r.recipe_id IN (%s)
|
||||||
|
AND %s
|
||||||
|
AND l.year IN (%s)",
|
||||||
|
.sql_lit_chr(candidates), .cohort_sql(cohort, "l.canonical_govid"),
|
||||||
|
paste(gap_years, collapse = ",")
|
||||||
|
))
|
||||||
|
res$year <- as.integer(res$year)
|
||||||
|
res
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Query recipe metadata: labels and year spans for a set of candidate
|
||||||
|
#' recipes.
|
||||||
|
#'
|
||||||
|
#' @param con Active DuckDB connection.
|
||||||
|
#' @param candidates Character vector of recipe IDs to look up.
|
||||||
|
#' @return Tibble with columns `recipe_id`, `label`, `year_min` (int), and
|
||||||
|
#' `year_max` (int).
|
||||||
|
#' @noRd
|
||||||
|
.query_recipe_meta <- function(con, candidates) {
|
||||||
|
tibble::as_tibble(DBI::dbGetQuery(con, sprintf(
|
||||||
|
"SELECT recipe_id, any_value(label) AS label,
|
||||||
|
MIN(year_min) AS year_min, MAX(year_max) AS year_max
|
||||||
|
FROM harmonization_recipes
|
||||||
|
WHERE recipe_id IN (%s)
|
||||||
|
GROUP BY recipe_id",
|
||||||
|
.sql_lit_chr(candidates)
|
||||||
|
)))
|
||||||
|
}
|
||||||
|
|
||||||
#' Attach `ig_recipe_id` to each suggestion: the intergovernmental-expenditure
|
#' Attach `ig_recipe_id` to each suggestion: the intergovernmental-expenditure
|
||||||
#' recipe (an M-to-local or L-to-state recipe) whose component codes cover
|
#' recipe (an M-to-local or L-to-state recipe) whose component codes cover
|
||||||
#' exactly the same set of function suffixes as the firing recipe's own
|
#' exactly the same set of function suffixes as the firing recipe's own
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
# uscogdata
|
# uscogdata
|
||||||
|
|
||||||
<!-- badges: start -->
|
<!-- badges: start -->
|
||||||
|
[](https://github.com/civilytics/uscogdata/actions/workflows/R-CMD-check.yaml)
|
||||||
[](https://civilytics.r-universe.dev/uscogdata)
|
[](https://civilytics.r-universe.dev/uscogdata)
|
||||||
[](LICENSE.md)
|
[](LICENSE.md)
|
||||||
<!-- badges: end -->
|
<!-- badges: end -->
|
||||||
@@ -226,7 +227,8 @@ A statewide total resting on a fifth of the universe looks exactly like one
|
|||||||
resting on all of it, so every multi-government result now says which it is:
|
resting on all of it, so every multi-government result now says which it is:
|
||||||
|
|
||||||
```r
|
```r
|
||||||
attr(rollup, "provenance")$coverage # per-year n_units_reporting, is_census_year
|
attr(rollup, "provenance")$coverage
|
||||||
|
# per-year n_units_expected, n_units_collected, n_units_reporting, is_census_year
|
||||||
```
|
```
|
||||||
|
|
||||||
`cog_geographic_rollup()`, `cog_peer_compare()` and `cog_find_peers()` take a
|
`cog_geographic_rollup()`, `cog_peer_compare()` and `cog_find_peers()` take a
|
||||||
@@ -234,8 +236,14 @@ attr(rollup, "provenance")$coverage # per-year n_units_reporting, is_census_ye
|
|||||||
`"consistent"` (only units reporting in every requested year, a balanced
|
`"consistent"` (only units reporting in every requested year, a balanced
|
||||||
panel).
|
panel).
|
||||||
|
|
||||||
`n_units_reporting` is **category-conditional**, and it is not a response rate. A government that was surveyed and genuinely spends
|
`n_units_reporting` is **category-conditional**: it counts governments with
|
||||||
nothing in the requested category is indistinguishable from one never surveyed.
|
rows for the *specific* category you asked for, so a government that was
|
||||||
|
surveyed and genuinely spends nothing in that category is indistinguishable
|
||||||
|
from one never surveyed — it is not a response rate on its own.
|
||||||
|
`n_units_collected` is the number that separates them: governments present in
|
||||||
|
the corpus that year for *any* category. `n_units_collected / n_units_expected`
|
||||||
|
is the true collection rate; `n_units_reporting / n_units_collected` is
|
||||||
|
category participation among collected units.
|
||||||
|
|
||||||
### Absent cells mean two different things
|
### Absent cells mean two different things
|
||||||
|
|
||||||
|
|||||||
@@ -55,11 +55,13 @@ Direct spending); `"primary"` and `"direct"` combine safely.}
|
|||||||
a balanced panel.
|
a balanced panel.
|
||||||
|
|
||||||
Regardless of mode, `provenance$coverage` always carries per-year
|
Regardless of mode, `provenance$coverage` always carries per-year
|
||||||
`n_units_reporting`, `n_units_expected` and `is_census_year`, and
|
`n_units_expected`, `n_units_collected`, `n_units_reporting` and
|
||||||
`provenance$coverage_mode` records the mode. `is_census_year` is a
|
`is_census_year`, and `provenance$coverage_mode` records the mode.
|
||||||
statement about the **survey calendar**, never a claim of completeness:
|
`is_census_year` is a statement about the **survey calendar**, never a
|
||||||
FY1967 is a census year in which only 97 of Wisconsin's 608 cities
|
claim of completeness: FY1967 is a census year in which only 97 of
|
||||||
report. `n_units_reporting` is the number that tells the truth.}
|
Wisconsin's 608 cities report. `n_units_reporting` is
|
||||||
|
category-conditional and is not a response rate on its own -- see
|
||||||
|
"Reading `coverage`" below for what each counter answers.}
|
||||||
}
|
}
|
||||||
\value{
|
\value{
|
||||||
Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`,
|
Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`,
|
||||||
@@ -86,14 +88,23 @@ by design — see `vignette('population-denominators')`.
|
|||||||
}
|
}
|
||||||
\section{Reading `coverage`}{
|
\section{Reading `coverage`}{
|
||||||
|
|
||||||
`provenance$coverage` reports `n_units_reporting` against
|
`provenance$coverage` carries three per-year counters:
|
||||||
`n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
|
||||||
it counts governments with rows for the category you asked for, not
|
* `n_units_expected` -- how many governments you asked about.
|
||||||
governments collected that year.** A government that was surveyed and
|
* `n_units_collected` -- how many of those appear in the corpus at all
|
||||||
genuinely spends nothing in that category is indistinguishable here from one
|
that year (in ANY category), separating sampling from real zeros.
|
||||||
that was never surveyed.
|
* `n_units_reporting` -- how many have rows for the SPECIFIC category you
|
||||||
|
requested. This is always <= n_units_collected: a government can be
|
||||||
|
collected but have no rows for "Police" because it contracts policing
|
||||||
|
to the county sheriff, not because it wasn't surveyed.
|
||||||
|
|
||||||
|
**`n_units_reporting` is category-conditional** and therefore **not a
|
||||||
|
response rate**: `n_units_reporting / n_units_expected` conflates sampling
|
||||||
|
(never collected) with real zeros (collected but spends nothing in your
|
||||||
|
category). Use `n_units_collected / n_units_expected` for the true
|
||||||
|
collection rate, and `n_units_reporting / n_units_collected` for category
|
||||||
|
participation among collected units.
|
||||||
|
|
||||||
The ratio is therefore **not a response rate** and must not be used as one.
|
|
||||||
In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||||
`category = "Police"`; the 174-city gap is overwhelmingly cities that
|
`category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||||
contract policing to the county sheriff, not non-response.
|
contract policing to the county sheriff, not non-response.
|
||||||
|
|||||||
+23
-12
@@ -50,11 +50,13 @@ safely.}
|
|||||||
a balanced panel.
|
a balanced panel.
|
||||||
|
|
||||||
Regardless of mode, `provenance$coverage` always carries per-year
|
Regardless of mode, `provenance$coverage` always carries per-year
|
||||||
`n_units_reporting`, `n_units_expected` and `is_census_year`, and
|
`n_units_expected`, `n_units_collected`, `n_units_reporting` and
|
||||||
`provenance$coverage_mode` records the mode. `is_census_year` is a
|
`is_census_year`, and `provenance$coverage_mode` records the mode.
|
||||||
statement about the **survey calendar**, never a claim of completeness:
|
`is_census_year` is a statement about the **survey calendar**, never a
|
||||||
FY1967 is a census year in which only 97 of Wisconsin's 608 cities
|
claim of completeness: FY1967 is a census year in which only 97 of
|
||||||
report. `n_units_reporting` is the number that tells the truth.
|
Wisconsin's 608 cities report. `n_units_reporting` is
|
||||||
|
category-conditional and is not a response rate on its own -- see
|
||||||
|
"Reading `coverage`" below for what each counter answers.
|
||||||
|
|
||||||
The comparison target is exempt from `"consistent"` balancing -- it is the
|
The comparison target is exempt from `"consistent"` balancing -- it is the
|
||||||
subject of the comparison, not a member of the cohort -- and the
|
subject of the comparison, not a member of the cohort -- and the
|
||||||
@@ -109,14 +111,23 @@ them.
|
|||||||
}
|
}
|
||||||
\section{Reading `coverage`}{
|
\section{Reading `coverage`}{
|
||||||
|
|
||||||
`provenance$coverage` reports `n_units_reporting` against
|
`provenance$coverage` carries three per-year counters:
|
||||||
`n_units_expected` per year. **`n_units_reporting` is category-conditional:
|
|
||||||
it counts cohort members with rows for the category you asked for, not
|
* `n_units_expected` -- how many governments you asked about.
|
||||||
cohort members collected that year.** A government that was surveyed and
|
* `n_units_collected` -- how many of those appear in the corpus at all
|
||||||
genuinely spends nothing in that category is indistinguishable here from one
|
that year (in ANY category), separating sampling from real zeros.
|
||||||
that was never surveyed.
|
* `n_units_reporting` -- how many have rows for the SPECIFIC category you
|
||||||
|
requested. This is always <= n_units_collected: a government can be
|
||||||
|
collected but have no rows for "Police" because it contracts policing
|
||||||
|
to the county sheriff, not because it wasn't surveyed.
|
||||||
|
|
||||||
|
**`n_units_reporting` is category-conditional** and therefore **not a
|
||||||
|
response rate**: `n_units_reporting / n_units_expected` conflates sampling
|
||||||
|
(never collected) with real zeros (collected but spends nothing in your
|
||||||
|
category). Use `n_units_collected / n_units_expected` for the true
|
||||||
|
collection rate, and `n_units_reporting / n_units_collected` for category
|
||||||
|
participation among collected units.
|
||||||
|
|
||||||
The ratio is therefore **not a response rate** and must not be used as one.
|
|
||||||
In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
In FY2022 — a complete census year — Georgia reports 393 of 567 cities for
|
||||||
`category = "Police"`; the 174-city gap is overwhelmingly cities that
|
`category = "Police"`; the 174-city gap is overwhelmingly cities that
|
||||||
contract policing to the county sheriff, not non-response.
|
contract policing to the county sheriff, not non-response.
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,51 @@
|
|||||||
|
# Project journal
|
||||||
|
|
||||||
|
Append-only, newest first. **Entries are never edited** — the value of this file is
|
||||||
|
that it records what was believed at the time, including the parts that turned out
|
||||||
|
wrong. Where things stand *today* is in `STATUS.md`, which is generated.
|
||||||
|
|
||||||
|
Four lines per entry. The analysis belongs in the issue or the decision record; this
|
||||||
|
file carries the reasoning and the pointers.
|
||||||
|
|
||||||
|
- **Why** — the driver. The one line git cannot reconstruct later.
|
||||||
|
- **Obligates** — issues this change created elsewhere. Numbers, not prose.
|
||||||
|
- **Refs** — commits, issues, decision records.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 2026-09-30 · repo · moved out of Nextcloud; code syncs only through Gitea
|
||||||
|
|
||||||
|
**Why:** Nextcloud was syncing this repo's `.git`, which risks conflict copies inside
|
||||||
|
it and stalls the client's cold scan across dozens of repos. Nothing untracked is
|
||||||
|
data, so nothing is linked (`--no-data`); the move changed no test result (1,101
|
||||||
|
pass, 2 skip before and after).
|
||||||
|
**Obligates:** #73, #74
|
||||||
|
**Refs:** 4bedf85 (ci/apt-https pushed to clear the move's preflight), 63b421a
|
||||||
|
|
||||||
|
## 2026-09-09 · ci · Linux R-CMD-check legs unblocked (recorded 2026-09-30)
|
||||||
|
|
||||||
|
**Why:** GitHub's ubuntu-latest image carries a google-chrome apt source that
|
||||||
|
intermittently fails its own hash check, and setup-r's `apt-get update` died on it for
|
||||||
|
every R version; the source is now found by URL and removed, and the step fails
|
||||||
|
loudly if it survives.
|
||||||
|
**Obligates:** (none)
|
||||||
|
**Refs:** d83301b, d1165a5
|
||||||
|
|
||||||
|
## 2026-09-09 · model · #36 coverage counter finished, two bugs caught before merge
|
||||||
|
|
||||||
|
**Why:** the drafted n_units_collected fix (uncommitted) derived its candidate
|
||||||
|
cohort from category-filtered results instead of the caller's full expected
|
||||||
|
cohort, and hardcoded a view name absent below schema_version 5 -- both
|
||||||
|
silent on the fixture, both would have shipped without an independent review
|
||||||
|
pass before merge.
|
||||||
|
**Obligates:** #72
|
||||||
|
**Refs:** #36, b41d5ee, PR#71
|
||||||
|
|
||||||
|
## 2026-09-09 · model · #33/#34 split into independently-tested commits
|
||||||
|
|
||||||
|
**Why:** #33's own branch had its tests sitting uncommitted, and quietly bundled
|
||||||
|
a behavior change (#34) into what its commit message called a pure refactor;
|
||||||
|
splitting them let each pass CI with its own tests instead of merging on a
|
||||||
|
false "tests pass" claim.
|
||||||
|
**Obligates:** (none)
|
||||||
|
**Refs:** #33, #34, 0c7c7eb, 392643b, d0d724c, 9f5cd98, PR#69, PR#70
|
||||||
@@ -0,0 +1,51 @@
|
|||||||
|
# Project status
|
||||||
|
|
||||||
|
> Between the compass markers is generated. Edit the sources, not this.
|
||||||
|
|
||||||
|
<!-- compass:begin -->
|
||||||
|
<!-- compass:board -->
|
||||||
|
|
||||||
|
## Where this stands
|
||||||
|
|
||||||
|
uscogdata's code now lives only in Gitea; the old Nextcloud folder keeps just build
|
||||||
|
output and working files. The move changed no test result: 1,101 pass and the two
|
||||||
|
opt-in live-corpus tests skip, before and after. It turned up one defect, #73: the
|
||||||
|
fixture's reference docs have never been in git, because a `.gitignore` rule matches
|
||||||
|
too broadly. Next in line are #73 and #72 (a vignette for the coverage counters), with
|
||||||
|
#52 (data-correction intake) and #74 (the stale state section in `CLAUDE.md`) waiting
|
||||||
|
on a decision.
|
||||||
|
|
||||||
|
## Ready to work on next
|
||||||
|
|
||||||
|
- **#73** The fixture corpus's docs/ never reaches git: .gitignore's docs/ rule is unanchored · `ws/corpus` — nothing is blocking it; something is currently wrong
|
||||||
|
- **#72** docs: add a vignette explaining provenance$coverage counters · `ws/docs` — nothing is blocking it; owed work from an earlier change
|
||||||
|
- **#2** Extend population data to be households as an alternate spending denominator · `ws/corpus` — nothing is blocking it
|
||||||
|
- **#64** Partition-level caching: R/cache.R is still a stub, and the remote path pays for it every session · `ws/corpus` — nothing is blocking it
|
||||||
|
- **#74** CLAUDE.md's Current State section is frozen at 2026-08-03 · `ws/docs` — waiting on a person, not on other work
|
||||||
|
- **#52** Release 11/11: design the data-correction intake (deferred; gates the API announcement) · `ws/corpus` — waiting on a person, not on other work
|
||||||
|
|
||||||
|
## Workstreams
|
||||||
|
|
||||||
|
| Stream | Commits since | Open | Debt | Owes docs |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| Query verbs and results | 0 | 0 | 0 | no |
|
||||||
|
| Corpus, mirror, provenance | 0 | 4 | 0 | no |
|
||||||
|
| Vignettes and guides | 0 | 2 | 2 | no |
|
||||||
|
|
||||||
|
## CI
|
||||||
|
|
||||||
|

|
||||||
|

|
||||||
|
|
||||||
|
<details>
|
||||||
|
<summary>Dependency graph and detail</summary>
|
||||||
|
|
||||||
|
_Nothing blocks anything else, so there is no graph to draw._
|
||||||
|
|
||||||
|
- Marker: `9adb9211` (2026-09-09)
|
||||||
|
- Commits since: 3
|
||||||
|
- Open issues: 6
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<!-- compass:end -->
|
||||||
@@ -0,0 +1,44 @@
|
|||||||
|
[project]
|
||||||
|
name = "uscogdata"
|
||||||
|
forge = "Civilytics/uscogdata"
|
||||||
|
|
||||||
|
# Three strands that go stale independently: what the verbs return, what the
|
||||||
|
# corpus is and how it is mounted, and how both are explained to a reader.
|
||||||
|
|
||||||
|
[[workstream]]
|
||||||
|
id = "api"
|
||||||
|
title = "Query verbs and results"
|
||||||
|
paths = [
|
||||||
|
"R/revenue.R", "R/spending.R", "R/balances.R", "R/peers.R", "R/search.R",
|
||||||
|
"R/categories.R", "R/recipes.R", "R/rollup.R", "R/explain.R", "R/basket.R",
|
||||||
|
"R/suggestions.R", "R/suppression.R", "R/complete.R", "R/cohort.R",
|
||||||
|
"R/basis.R", "R/adjust.R", "R/pagination.R",
|
||||||
|
]
|
||||||
|
docs = ["vignettes/*.Rmd", "README.md"]
|
||||||
|
|
||||||
|
[[workstream]]
|
||||||
|
id = "corpus"
|
||||||
|
title = "Corpus, mirror, provenance"
|
||||||
|
paths = [
|
||||||
|
"R/manifest.R", "R/mirror.R", "R/cache.R", "R/session.R", "R/provenance.R",
|
||||||
|
"R/coverage.R", "R/config.R", "R/views.R", "R/series_breaks.R",
|
||||||
|
"R/balance_caveats.R", "R/zzz.R", "data-raw/**", "inst/sql/**",
|
||||||
|
]
|
||||||
|
docs = ["vignettes/*.Rmd", "NEWS.md"]
|
||||||
|
|
||||||
|
[[workstream]]
|
||||||
|
id = "docs"
|
||||||
|
title = "Vignettes and guides"
|
||||||
|
paths = ["vignettes/**", "README.md", "_pkgdown.yml", "NEWS.md"]
|
||||||
|
docs = []
|
||||||
|
|
||||||
|
[roborev]
|
||||||
|
project_guidelines = [
|
||||||
|
"Every verb calls .ensure_session() first, then queries via DBI::dbGetQuery().",
|
||||||
|
"A verb's return value is always a tbl_df carrying a provenance attribute.",
|
||||||
|
"govid inputs always go through .coerce_govid_input(); it accepts a character vector or a data frame.",
|
||||||
|
"SQL has two layers: view definitions are numbered .sql files in inst/sql/ registered by .register_views(); query construction is inline sprintf() in R. Add a view as a file; build a query in R.",
|
||||||
|
"No arrow dependency -- DuckDB reads parquet natively.",
|
||||||
|
"withr is Suggests-only and must appear in tests alone.",
|
||||||
|
"Tests must pass offline against the bundled fixture; tests/testthat/setup.R sets USCOGDATA_URL for that.",
|
||||||
|
]
|
||||||
@@ -0,0 +1,12 @@
|
|||||||
|
# Decisions
|
||||||
|
|
||||||
|
One file per decision, numbered and immutable. A decision that changes is superseded
|
||||||
|
by a new record, never edited in place — the old reasoning is the point.
|
||||||
|
|
||||||
|
The table below is **generated** by `compass:decide`. Do not hand-edit it.
|
||||||
|
|
||||||
|
<!-- compass:begin decisions -->
|
||||||
|
| # | Date | Decision | Status |
|
||||||
|
|---|---|---|---|
|
||||||
|
| — | — | *No decisions recorded yet.* | — |
|
||||||
|
<!-- compass:end decisions -->
|
||||||
@@ -48,6 +48,13 @@ test_that("multi-government aggregates disclose reporting coverage on every resu
|
|||||||
expect_equal(cov$n_units_reporting, c(152L, 597L, 112L, 114L))
|
expect_equal(cov$n_units_reporting, c(152L, 597L, 112L, 114L))
|
||||||
expect_equal(cov$is_census_year, c(FALSE, TRUE, FALSE, FALSE))
|
expect_equal(cov$is_census_year, c(FALSE, TRUE, FALSE, FALSE))
|
||||||
|
|
||||||
|
# uscogdata#36: with category = NULL (no category scope), "reported at
|
||||||
|
# all" and "collected" are the same question, so n_units_collected must
|
||||||
|
# equal n_units_reporting exactly here. This case alone cannot catch a
|
||||||
|
# regression in HOW n_units_collected is computed, though: see the
|
||||||
|
# category-scoped test below for that.
|
||||||
|
expect_equal(cov$n_units_collected, cov$n_units_reporting)
|
||||||
|
|
||||||
# Cross-check against the raw partitions, scoped to the SAME universe the
|
# Cross-check against the raw partitions, scoped to the SAME universe the
|
||||||
# rollup was given -- the 608 govids above. Scoping instead on the long
|
# rollup was given -- the 608 govids above. Scoping instead on the long
|
||||||
# table's own `type`/`fips_state` asks a different question and answers 595:
|
# table's own `type`/`fips_state` asks a different question and answers 595:
|
||||||
@@ -80,6 +87,8 @@ test_that("multi-government aggregates disclose reporting coverage on every resu
|
|||||||
expect_equal(cov_peers$n_units_expected, rep(15L, 3L))
|
expect_equal(cov_peers$n_units_expected, rep(15L, 3L))
|
||||||
expect_equal(cov_peers$n_units_reporting, c(15L, 3L, 3L))
|
expect_equal(cov_peers$n_units_reporting, c(15L, 3L, 3L))
|
||||||
expect_equal(cov_peers$is_census_year, c(TRUE, FALSE, FALSE))
|
expect_equal(cov_peers$is_census_year, c(TRUE, FALSE, FALSE))
|
||||||
|
# uscogdata#36: same identity as the rollup case above, category = NULL.
|
||||||
|
expect_equal(cov_peers$n_units_collected, cov_peers$n_units_reporting)
|
||||||
|
|
||||||
# -- the three coverage modes --------------------------------------------
|
# -- the three coverage modes --------------------------------------------
|
||||||
expect_equal(attr(cog_peer_compare(target_govid = chilton, peers = peers,
|
expect_equal(attr(cog_peer_compare(target_govid = chilton, peers = peers,
|
||||||
@@ -101,3 +110,102 @@ test_that("multi-government aggregates disclose reporting coverage on every resu
|
|||||||
coverage = "census")
|
coverage = "census")
|
||||||
expect_equal(sort(unique(census_only$year)), 2012)
|
expect_equal(sort(unique(census_only$year)), 2012)
|
||||||
})
|
})
|
||||||
|
|
||||||
|
test_that("n_units_collected separates sampling from real zeros, category-scoped (uscogdata#36)", {
|
||||||
|
# The motivating case from the issue: Wisconsin cities, category = "Police".
|
||||||
|
# FY2012 is a complete census year -- collection is not partial -- yet a
|
||||||
|
# category-conditional n_units_reporting alone reads like a sampling gap.
|
||||||
|
# n_units_collected must diverge from n_units_reporting here, unlike the
|
||||||
|
# category = NULL cases above, because most of the FY2012 gap is cities
|
||||||
|
# that contract policing to the county sheriff (collected, real zero), not
|
||||||
|
# cities Census never surveyed.
|
||||||
|
wi <- cog_gov_search(name = NULL, state = "WI", type = "city")
|
||||||
|
roll <- suppressMessages(cog_geographic_rollup(
|
||||||
|
govids = list(city = wi$canonical_govid), category = "Police",
|
||||||
|
years = c(2011L, 2012L, 2019L, 2020L)))
|
||||||
|
cov <- wt_coverage(roll)
|
||||||
|
|
||||||
|
expect_equal(cov$n_units_expected, rep(608L, 4L))
|
||||||
|
expect_equal(cov$n_units_collected, c(152L, 597L, 112L, 114L))
|
||||||
|
expect_equal(cov$n_units_reporting, c(152L, 485L, 109L, 111L))
|
||||||
|
|
||||||
|
# The pair the issue actually wants: collected/expected is the true
|
||||||
|
# collection rate (98% in the FY2012 census year, matching the raw
|
||||||
|
# cross-check above); reporting/collected is category participation among
|
||||||
|
# collected units (81% -- most of the gap is real, not sampling).
|
||||||
|
expect_equal(round(cov$n_units_collected[cov$year == 2012] /
|
||||||
|
cov$n_units_expected[cov$year == 2012], 2), 0.98)
|
||||||
|
expect_equal(round(cov$n_units_reporting[cov$year == 2012] /
|
||||||
|
cov$n_units_collected[cov$year == 2012], 2), 0.81)
|
||||||
|
|
||||||
|
# Every year: collected is bounded between reporting and expected.
|
||||||
|
expect_true(all(cov$n_units_collected >= cov$n_units_reporting))
|
||||||
|
expect_true(all(cov$n_units_collected <= cov$n_units_expected))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".coverage_table() candidates a government collected-but-absent from the category result (uscogdata#36)", {
|
||||||
|
# Direct regression test for the mechanism itself: n_units_collected's
|
||||||
|
# candidate list must be the caller's full expected cohort (expected_ids),
|
||||||
|
# never derived from `result`/`rows`. A government with zero rows in the
|
||||||
|
# requested category across every requested year never appears in
|
||||||
|
# `result` at all, so deriving candidates from `result` would silently
|
||||||
|
# drop exactly the "collected but real zero" governments this counter
|
||||||
|
# exists to count -- collapsing it back to n_units_reporting.
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
|
||||||
|
# A real fixture govid, present in spending_long_harmonized for 2019 (in
|
||||||
|
# SOME category), but absent from this fake category-specific `result`.
|
||||||
|
govid <- "011021100004"
|
||||||
|
fake_result <- data.frame(canonical_govid = character(0), year = integer(0))
|
||||||
|
|
||||||
|
cov <- uscogdata:::.coverage_table(
|
||||||
|
fake_result, years = 2019L, n_expected = 1L,
|
||||||
|
con = con, long_view = "spending_long_harmonized",
|
||||||
|
expected_ids = govid
|
||||||
|
)
|
||||||
|
expect_equal(cov$n_units_collected, 1L)
|
||||||
|
expect_equal(cov$n_units_reporting, 0L)
|
||||||
|
|
||||||
|
# Without a connection, long_view, or expected_ids, the lookup is skipped
|
||||||
|
# rather than silently wrong.
|
||||||
|
no_con <- uscogdata:::.coverage_table(fake_result, years = 2019L, n_expected = 1L)
|
||||||
|
expect_true(is.na(no_con$n_units_collected))
|
||||||
|
|
||||||
|
no_ids <- uscogdata:::.coverage_table(
|
||||||
|
fake_result, years = 2019L, n_expected = 1L,
|
||||||
|
con = con, long_view = "spending_long_harmonized"
|
||||||
|
)
|
||||||
|
expect_true(is.na(no_ids$n_units_collected))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("n_units_collected uses the resolved basis's long view, not a hardcoded harmonized one (uscogdata#36)", {
|
||||||
|
# spending_long_harmonized only exists when schema_version >= 5 (R/views.R
|
||||||
|
# gates the harmonization views on it); on an older corpus cog_spending()
|
||||||
|
# resolves basis = "raw" and queries spending_long instead. The coverage
|
||||||
|
# lookup must follow the SAME resolved basis, not a literal
|
||||||
|
# "spending_long_harmonized", or it hard-errors with a DuckDB catalog
|
||||||
|
# error on every schema_version < 5 corpus -- a vintage the package
|
||||||
|
# otherwise explicitly still supports (see test-manifest.R's dual-accept
|
||||||
|
# tests).
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_doctored_schema_version(4L, {
|
||||||
|
con <- cog_open()
|
||||||
|
ids <- DBI::dbGetQuery(con,
|
||||||
|
"SELECT DISTINCT canonical_govid FROM spending_long WHERE year = 2011 LIMIT 3"
|
||||||
|
)$canonical_govid
|
||||||
|
expect_gte(length(ids), 3L)
|
||||||
|
|
||||||
|
roll <- suppressMessages(cog_geographic_rollup(
|
||||||
|
list(city = ids), category = NULL, years = 2011L))
|
||||||
|
expect_equal(attr(roll, "provenance")$basis, "raw")
|
||||||
|
cov <- attr(roll, "provenance")$coverage
|
||||||
|
expect_false(is.na(cov$n_units_collected))
|
||||||
|
expect_equal(cov$n_units_collected, length(ids))
|
||||||
|
|
||||||
|
cmp <- suppressMessages(cog_peer_compare(
|
||||||
|
target_govid = ids[1], peers = ids[-1], category = NULL, years = 2011L))
|
||||||
|
expect_equal(attr(cmp, "provenance")$basis, "raw")
|
||||||
|
cov_peers <- attr(cmp, "provenance")$coverage
|
||||||
|
expect_false(is.na(cov_peers$n_units_collected))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|||||||
@@ -315,15 +315,17 @@ test_that("a mis-scoped cog_spending() call never attaches an M/L counterpart to
|
|||||||
# (SB194, cog_pipeline#64), so the recipe stopped being a candidate there.
|
# (SB194, cog_pipeline#64), so the recipe stopped being a candidate there.
|
||||||
# FL state carries a real FY2011 B47 amount, so this exercises the guard
|
# FL state carries a real FY2011 B47 amount, so this exercises the guard
|
||||||
# against a suggestion that genuinely fires.
|
# against a suggestion that genuinely fires.
|
||||||
|
#
|
||||||
|
# Issue #34: "IG Federal" maps to B-prefixed codes in summary_categories
|
||||||
|
# with category_type = 'revenue'. A spending verb (flow_prefixes E/F/G)
|
||||||
|
# now scopes its candidate query by category_type = 'expenditure', so it
|
||||||
|
# correctly finds NO candidates for this revenue-only category -- the
|
||||||
|
# suggestion machinery cannot fire, and no M/L counterpart is attached.
|
||||||
r <- suppressMessages(
|
r <- suppressMessages(
|
||||||
cog_spending("120000226351", years = c(2005, 2011), category = "IG Federal")
|
cog_spending("120000226351", years = c(2005, 2011), category = "IG Federal")
|
||||||
)
|
)
|
||||||
sugg <- attr(r, "provenance")$suggestions
|
sugg <- attr(r, "provenance")$suggestions
|
||||||
expect_gt(length(sugg), 0L)
|
expect_length(sugg, 0L)
|
||||||
ids <- vapply(sugg, function(s) s$recipe_id %||% "", character(1))
|
|
||||||
expect_true("ig_federal_b47_wide" %in% ids)
|
|
||||||
ig <- unlist(lapply(sugg, function(s) s$ig_recipe_id))
|
|
||||||
expect_length(ig, 0L)
|
|
||||||
})
|
})
|
||||||
|
|
||||||
test_that("C1: 'total' on a legacy aggregate-only family reports the IG-only figure honestly, not as Direct + IG", {
|
test_that("C1: 'total' on a legacy aggregate-only family reports the IG-only figure honestly, not as Direct + IG", {
|
||||||
|
|||||||
@@ -118,3 +118,17 @@ test_that("cog_explain prints denominator + popyear_range + counts", {
|
|||||||
expect_false(grepl("popyear range: 19-20", out, fixed = TRUE))
|
expect_false(grepl("popyear range: 19-20", out, fixed = TRUE))
|
||||||
})
|
})
|
||||||
})
|
})
|
||||||
|
|
||||||
|
test_that("cog_explain reports units collected alongside units reporting (uscogdata#36)", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
wi <- cog_gov_search(name = NULL, state = "WI", type = "city")
|
||||||
|
roll <- suppressMessages(cog_geographic_rollup(
|
||||||
|
govids = list(city = wi$canonical_govid), category = "Police",
|
||||||
|
years = 2012L))
|
||||||
|
out <- paste(c(
|
||||||
|
capture.output(cog_explain(roll)),
|
||||||
|
capture.output(cog_explain(roll), type = "message")
|
||||||
|
), collapse = "\n")
|
||||||
|
expect_true(grepl("597 of 608 units collected", out, fixed = TRUE))
|
||||||
|
expect_true(grepl("485 reporting in this category", out, fixed = TRUE))
|
||||||
|
})
|
||||||
|
|||||||
@@ -371,34 +371,45 @@ test_that("uscogdata#9: the revenue verb inherits the same trigger", {
|
|||||||
expect_null(sugg[[1]]$ig_recipe_id)
|
expect_null(sugg[[1]]$ig_recipe_id)
|
||||||
})
|
})
|
||||||
|
|
||||||
test_that("I1: cog_revenue never fabricates suppressed dollars for an expenditure-only recipe", {
|
test_that("I1 + #34: cog_revenue never suggests expenditure-only recipes", {
|
||||||
# uscogdata#9 review, finding I1: Corrections is an expenditure-only
|
# uscogdata#9 review, finding I1: Corrections is an expenditure-only
|
||||||
# category (E04/E05). cog_revenue() naturally returns zero rows for it, so
|
# category (E04/E05). Before the flow_prefixes fix (#9), .suppressed_components()
|
||||||
# corrections_combined still fires as an empty_year suggestion (its own
|
# measured E04/E05 against cog_revenue()'s OWN view and reported $3.6B as
|
||||||
# generic join finds real E04/E05 data for this government) -- but before
|
# "suppressed" -- nothing was suppressed at all.
|
||||||
# the flow_prefixes fix, .suppressed_components() measured E04/E05 against
|
#
|
||||||
# cog_revenue()'s OWN view (which can never contain an E-coded row by
|
# Issue #34 builds on that: the candidate query now also filters by
|
||||||
# construction) and reported the full $3,631,945,000 as "suppressed",
|
# category_type ('revenue'), so expenditure-only recipes like corrections_combined
|
||||||
# when cog_spending() for the same gov/years/category actually returns
|
# (whose components E04/E05 are classified as 'expenditure' in summary_categories)
|
||||||
# $3,691,029,000 -- nothing was suppressed at all.
|
# are never even considered for a revenue verb. This is stronger than just
|
||||||
|
# suppressing the dollar claim -- it prevents the suggestion from firing at all.
|
||||||
skip_if_no_corpus()
|
skip_if_no_corpus()
|
||||||
r <- suppressMessages(
|
r <- suppressMessages(
|
||||||
cog_revenue("061037123085", years = 2019:2020, category = "Corrections"))
|
cog_revenue("061037123085", years = 2019:2020, category = "Corrections"))
|
||||||
sugg <- attr(r, "provenance")$suggestions
|
sugg <- attr(r, "provenance")$suggestions
|
||||||
ids <- vapply(sugg, function(s) s$recipe_id, character(1))
|
ids <- vapply(sugg, function(s) s$recipe_id %||% "", character(1))
|
||||||
expect_true("corrections_combined" %in% ids)
|
|
||||||
|
|
||||||
hit <- sugg[[which(ids == "corrections_combined")]]
|
# corrections_combined should NOT appear -- its components are expenditure-only.
|
||||||
expect_equal(hit$suppressed_amount, 0)
|
expect_false("corrections_combined" %in% ids)
|
||||||
expect_equal(hit$suppressed_years, integer(0))
|
})
|
||||||
expect_equal(hit$suppressed_codes, character(0))
|
|
||||||
|
|
||||||
# And cog_spending() for the identical gov/years/category is unaffected --
|
test_that(".query_candidate_recipes() scopes candidates by category_type (#34)", {
|
||||||
# it actually finds the E04/E05 dollars the buggy measurement claimed were
|
# Direct assertion on the mechanism the two tests above exercise
|
||||||
# excluded.
|
# end-to-end: corrections_combined's own components (E04/E05) are
|
||||||
sp <- suppressMessages(
|
# category_type = 'expenditure' in summary_categories, so an
|
||||||
cog_spending("061037123085", years = 2019:2020, category = "Corrections"))
|
# expenditure-flavored flow_prefixes call must surface it and a
|
||||||
expect_equal(sum(sp$amt_nominal), 3691029000)
|
# revenue-flavored one must not. This queries only summary_categories/
|
||||||
|
# harmonization_recipes (no government data), so it runs against the
|
||||||
|
# bundled fixture with no skip_if_no_corpus() needed.
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
|
||||||
|
expenditure <- uscogdata:::.query_candidate_recipes(
|
||||||
|
con, category = "Corrections", flow_prefixes = c("E", "F", "G"))
|
||||||
|
expect_true("corrections_combined" %in% expenditure)
|
||||||
|
|
||||||
|
revenue <- uscogdata:::.query_candidate_recipes(
|
||||||
|
con, category = "Corrections",
|
||||||
|
flow_prefixes = c("T", "A", "U", "B", "C", "D"))
|
||||||
|
expect_false("corrections_combined" %in% revenue)
|
||||||
})
|
})
|
||||||
|
|
||||||
test_that("uscogdata#9: no partial-coverage fire in a modern year", {
|
test_that("uscogdata#9: no partial-coverage fire in a modern year", {
|
||||||
|
|||||||
Reference in New Issue
Block a user