From 2e317d3a0f73fa27fc14271d63f7dbb000c70442 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Sun, 23 Aug 2026 16:17:20 -0400 Subject: [PATCH] chore: adopt compass for project tracking Three workstreams -- the query verbs, the corpus and how it is mounted, and the docs that explain both. Each can go stale independently, which is what the workstream boundary is for: a change to a verb obligates the vignettes, a change to the corpus obligates NEWS. All six open issues now carry ws/ and type/ labels, applied additively so #36 kept its existing kodor/, severity/, south-guide and verdict/ labels. The generated board is pinned as issue #68. .roborev.toml carries review guidelines composed from a shared baseline, the R package overlay, and this project's own conventions read out of CLAUDE.md: the ensure_session-then-dbGetQuery order, the tbl_df-with-provenance return contract, coerce_govid_input at the boundary, the two SQL layers, no arrow, and withr as tests-only. The .gitignore change is load-bearing. pkgdown output made docs/ ignored, which would have left every compass file untracked and unable to travel to another machine. Git cannot re-include anything beneath an excluded directory, so the rule had to list children instead. That forced an anchoring change: a bare "docs/" matches at any depth, "/docs/*" only at the root, so the fixture corpus docs directory needed an explicit rule to stay excluded as before. --- .gitignore | 15 ++++++++- .roborev.toml | 56 ++++++++++++++++++++++++++++++ docs/decisions/README.md | 12 +++++++ docs/pm/JOURNAL.md | 14 ++++++++ docs/pm/STATUS.md | 73 ++++++++++++++++++++++++++++++++++++++++ docs/pm/compass.toml | 44 ++++++++++++++++++++++++ 6 files changed, 213 insertions(+), 1 deletion(-) create mode 100644 .roborev.toml create mode 100644 docs/decisions/README.md create mode 100644 docs/pm/JOURNAL.md create mode 100644 docs/pm/STATUS.md create mode 100644 docs/pm/compass.toml diff --git a/.gitignore b/.gitignore index f03f1e7..ccbe719 100644 --- a/.gitignore +++ b/.gitignore @@ -4,7 +4,17 @@ .Ruserdata *.Rproj inst/doc -docs/ +# pkgdown output. Listed as children rather than `docs/` so compass's +# docs/pm/ and docs/decisions/ can be re-included -- git cannot re-include +# anything beneath an excluded directory. +# +# Note the anchoring change this forces: a bare `docs/` matches a directory of +# that name at ANY depth, while `/docs/*` matches only at the repo root. The +# fixture corpus's own docs/ therefore needs its own rule to stay excluded. +/docs/* +!/docs/pm/ +!/docs/decisions/ +inst/extdata/fixture_corpus/docs/ /doc/ /Meta/ .DS_Store @@ -12,3 +22,6 @@ docs/ # SDD working artifacts (ledger, briefs, review packages) — plans/ stays tracked .superpowers/sdd/ +.compass-cache/ +# roborev snapshots +/.roborev/ diff --git a/.roborev.toml b/.roborev.toml new file mode 100644 index 0000000..29d690e --- /dev/null +++ b/.roborev.toml @@ -0,0 +1,56 @@ +# roborev configuration, initialised by compass. +# Reviews are queued to a background daemon -- they never block a commit. + +post_commit_review = 'commit' +excluded_commit_patterns = ['WIP', 'chore:', 'docs:', 'Merge '] + +review_guidelines = ''' +# --- compass:begin (generated -- edit the sources, not this) --- +- Prefer returning new values to mutating arguments in place. A function that edits + its caller's object is a bug waiting for a second caller. +- Validate at system boundaries -- user input, API responses, file contents, config. + Fail fast with a message naming the field and the file. +- Never swallow an error. Handle it or let it propagate; a bare catch that continues + is worse than a crash. +- No hardcoded secrets, tokens, or credentials, and no secrets in log output or error + messages. +- Parameterise every query. String-built SQL is a defect even when the input looks safe. +- Keep functions under roughly 50 lines and files under roughly 400. Flag nesting + deeper than four levels. +- No magic numbers or hardcoded paths -- name them as constants or read them from config. +- New behaviour needs a test. A bug fix needs a test that fails without the fix. +- Use the native pipe `|>`, not magrittr `%>%`. +- snake_case for objects and functions; UPPER_SNAKE for constants. Never use `.` as a + word separator in a function name -- it collides with S3 dispatch. +- Validate arguments at the top of exported functions with `stopifnot()` or an explicit + check, and say which argument was wrong. +- Never `setDT()`, `set()`, or otherwise modify by reference a data.table the caller + still owns. `as.data.table()` copies; use it. +- Prefer `vapply()` to `sapply()` -- `sapply()` silently returns a list when the type + varies, which turns a type error into a downstream mystery. +- Use `seq_len(n)` / `seq_along(x)`, never `1:n`, which iterates backwards when n is 0. +- Compare strings with `==` only after checking for NA; use `identical()` for scalars + where NA would be wrong. +- Do not call `library()` inside package or module files; attach packages in scripts and + test helpers only. +- Namespace-qualify calls into other packages (`stats::sd`) in code that is sourced. +- Every exported function needs roxygen with `@param` for each argument (type, meaning, + and why the default is what it is) and `@return`. Add `@examples` for exported API. +- Declare dependencies in DESCRIPTION. Prefer base R or an existing dependency over + adding a new one; a package with zero hard deps is worth keeping that way. +- Signal errors with `stop()` carrying a condition class, so callers can catch the kind + rather than matching on message text. +- Keep internals internal. Export only what a user needs; an accidentally exported + helper becomes an API you have to keep. +- Tests use testthat edition 3. Each test is self-sufficient -- no reliance on state + left by an earlier test or on a fixture built elsewhere in the file. +- Prefer duplication in tests over a helper that hides what is being asserted. +- Every verb calls .ensure_session() first, then queries via DBI::dbGetQuery(). +- A verb's return value is always a tbl_df carrying a provenance attribute. +- govid inputs always go through .coerce_govid_input(); it accepts a character vector or a data frame. +- SQL has two layers: view definitions are numbered .sql files in inst/sql/ registered by .register_views(); query construction is inline sprintf() in R. Add a view as a file; build a query in R. +- No arrow dependency -- DuckDB reads parquet natively. +- withr is Suggests-only and must appear in tests alone. +- Tests must pass offline against the bundled fixture; tests/testthat/setup.R sets USCOGDATA_URL for that. +# --- compass:end --- +''' diff --git a/docs/decisions/README.md b/docs/decisions/README.md new file mode 100644 index 0000000..62f0eed --- /dev/null +++ b/docs/decisions/README.md @@ -0,0 +1,12 @@ +# Decisions + +One file per decision, numbered and immutable. A decision that changes is superseded +by a new record, never edited in place — the old reasoning is the point. + +The table below is **generated** by `compass:decide`. Do not hand-edit it. + + +| # | Date | Decision | Status | +|---|---|---|---| +| — | — | *No decisions recorded yet.* | — | + diff --git a/docs/pm/JOURNAL.md b/docs/pm/JOURNAL.md new file mode 100644 index 0000000..b34a322 --- /dev/null +++ b/docs/pm/JOURNAL.md @@ -0,0 +1,14 @@ +# Project journal + +Append-only, newest first. **Entries are never edited** — the value of this file is +that it records what was believed at the time, including the parts that turned out +wrong. Where things stand *today* is in `STATUS.md`, which is generated. + +Four lines per entry. The analysis belongs in the issue or the decision record; this +file carries the reasoning and the pointers. + +- **Why** — the driver. The one line git cannot reconstruct later. +- **Obligates** — issues this change created elsewhere. Numbers, not prose. +- **Refs** — commits, issues, decision records. + +--- diff --git a/docs/pm/STATUS.md b/docs/pm/STATUS.md new file mode 100644 index 0000000..b77b06b --- /dev/null +++ b/docs/pm/STATUS.md @@ -0,0 +1,73 @@ +# Project status + +> Between the compass markers is generated. Edit the sources, not this. + + + + +## Where this stands + +uscogdata is at 0.4.0 and its public surface is settled: the query verbs, the cohort +predicates added in this release, and the provenance contract every verb returns. + +The six open issues split cleanly. Two are API work carried out of the #9 review pass +and deliberately deferred there rather than fixed in that branch. Three concern the +corpus layer, and the largest of them, partition-level caching, was named the single +highest-leverage change on the remote path before being deferred. One, the +data-correction intake, is a decision rather than a task: it was parked during the +0.3.0 design and it gates the API announcement, because without it the corpus cannot +make the "traceable and correctable" claim that most distinguishes it from Census's +own files. + +Nothing here is blocked on anything else, so the ordering is a judgement about value +rather than a dependency graph. + +## Ready to work on next + +- **#34** cog_revenue() offers expenditure recipes as suggestions: scope the candidate query by category_type · `ws/api` — nothing is blocking it; something is currently wrong +- **#36** n_units_reporting is category-conditional and cannot be read as a response rate · `ws/corpus` — nothing is blocking it; owed work from an earlier change +- **#2** Extend population data to be households as an alternate spending denominator · `ws/corpus` — nothing is blocking it +- **#33** Decompose .build_suggestions() (106 lines) into named helpers · `ws/api` — nothing is blocking it +- **#52** Release 11/11: design the data-correction intake (deferred; gates the API announcement) · `ws/corpus` — nothing is blocking it +- **#64** Partition-level caching: R/cache.R is still a stub, and the remote path pays for it every session · `ws/corpus` — nothing is blocking it + +## Workstreams + +| Stream | Commits since | Open | Debt | Owes docs | +|---|---|---|---|---| +| Query verbs and results | 77 | 2 | 0 | no | +| Corpus, mirror, provenance | 39 | 4 | 1 | no | +| Vignettes and guides | 34 | 0 | 0 | **yes** | + +## CI + +![R-CMD-check](https://gitea.civilytics.org/Civilytics/uscogdata/actions/workflows/ci.yml/badge.svg?branch=main) +![Mirror to GitHub](https://gitea.civilytics.org/Civilytics/uscogdata/actions/workflows/mirror-github.yml/badge.svg?branch=main) + +
+Dependency graph and detail + +```mermaid +graph TD + I34["#34 cog_revenue() offers expenditure recipes as sug…"] + I36["#36 n_units_reporting is category-conditional and c…"] + I2["#2 Extend population data to be households as an a…"] + I33["#33 Decompose .build_suggestions() (106 lines) into…"] + I52["#52 Release 11/11: design the data-correction intak…"] + I64["#64 Partition-level caching: R/cache.R is still a s…"] + class I34 ready; + class I36 ready; + class I2 ready; + class I33 ready; + class I52 ready; + class I64 ready; + classDef ready fill:#dafbe1,stroke:#2da44e; +``` + +- Marker: `none` (no journal entry yet) +- Commits since: 164 +- Open issues: 6 + +
+ + diff --git a/docs/pm/compass.toml b/docs/pm/compass.toml new file mode 100644 index 0000000..88c9e03 --- /dev/null +++ b/docs/pm/compass.toml @@ -0,0 +1,44 @@ +[project] +name = "uscogdata" +forge = "Civilytics/uscogdata" + +# Three strands that go stale independently: what the verbs return, what the +# corpus is and how it is mounted, and how both are explained to a reader. + +[[workstream]] +id = "api" +title = "Query verbs and results" +paths = [ + "R/revenue.R", "R/spending.R", "R/balances.R", "R/peers.R", "R/search.R", + "R/categories.R", "R/recipes.R", "R/rollup.R", "R/explain.R", "R/basket.R", + "R/suggestions.R", "R/suppression.R", "R/complete.R", "R/cohort.R", + "R/basis.R", "R/adjust.R", "R/pagination.R", +] +docs = ["vignettes/*.Rmd", "README.md"] + +[[workstream]] +id = "corpus" +title = "Corpus, mirror, provenance" +paths = [ + "R/manifest.R", "R/mirror.R", "R/cache.R", "R/session.R", "R/provenance.R", + "R/coverage.R", "R/config.R", "R/views.R", "R/series_breaks.R", + "R/balance_caveats.R", "R/zzz.R", "data-raw/**", "inst/sql/**", +] +docs = ["vignettes/*.Rmd", "NEWS.md"] + +[[workstream]] +id = "docs" +title = "Vignettes and guides" +paths = ["vignettes/**", "README.md", "_pkgdown.yml", "NEWS.md"] +docs = [] + +[roborev] +project_guidelines = [ + "Every verb calls .ensure_session() first, then queries via DBI::dbGetQuery().", + "A verb's return value is always a tbl_df carrying a provenance attribute.", + "govid inputs always go through .coerce_govid_input(); it accepts a character vector or a data frame.", + "SQL has two layers: view definitions are numbered .sql files in inst/sql/ registered by .register_views(); query construction is inline sprintf() in R. Add a view as a file; build a query in R.", + "No arrow dependency -- DuckDB reads parquet natively.", + "withr is Suggests-only and must appear in tests alone.", + "Tests must pass offline against the bundled fixture; tests/testthat/setup.R sets USCOGDATA_URL for that.", +]