Tracks the corpus published 2026-07-30, which adds category_type = 'balance' (pipeline#76) and the I/Q/Y flow codes (pipeline#78) -- the crosswalk prerequisite for #11's three-concept expenditure model. Fixture crosswalk goes 291 -> 324 rows and gains balance_subtype. Only three files change (series_breaks, summary_categories, manifest); no long partition moves, because the published change was metadata-only. test-categories.R's vocabulary assertions extended for the new values: category_type gains 'balance', spending subtypes gain 'interest' and 'insurance_benefits', revenue subtypes gain 'insurance_trust'. cog_categories() is a catalogue verb so it surfaces every category_type the corpus carries; the stock/flow guard belongs on the money verbs. Suite: 0 failures, 2 skips (the #11 and #12 blocks).
121 lines
5.2 KiB
JSON
121 lines
5.2 KiB
JSON
{
|
|
"schema_version": 6,
|
|
"built_at": "2026-07-30T20:01:56Z",
|
|
"pipeline_commit": "e64a046",
|
|
"fixture_note": "Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated from the sparsified schema-v6 corpus: the wide era (<= FY2011) no longer stores explicit zeros, so FY2011 absence means Census published $0 while FY2012+ absence means not reported. representation.parquet and code_set.parquet carry that rule and ship in full, as do every other metadata table in the publish tree. 2011/2012 straddle both the wide-aggregate -> modern-leaf format boundary (exercised by basis=\"harmonized\" and recipe= queries) and the dense -> sparse representation boundary (SB194); 2019/2020 retain the prior per-capita/CPI regression anchors. Regenerated via data-raw/regenerate_fixture_corpus.R.",
|
|
"data_vintage": {
|
|
"source_vintages": {
|
|
"2012": "10162019",
|
|
"2013": "10162019",
|
|
"2014": "10162019",
|
|
"2015": "10162019",
|
|
"2016": "10162019",
|
|
"2017": "06102021",
|
|
"2018": "06102021",
|
|
"2019": "06102021",
|
|
"2020": "06122023",
|
|
"2021": "06122023",
|
|
"2022": "06052025",
|
|
"2023": "06052025"
|
|
},
|
|
"registry_rows": 148,
|
|
"acs_vintage": "ACS 2018-2022 5-year"
|
|
},
|
|
"scope": {
|
|
"gov_types_included": [0, 1, 2, 3],
|
|
"gov_types_excluded": [4, 5],
|
|
"scope_note": "v0.1 covers state, county, city/municipality, and township governments. Special districts (type 4) and school districts (type 5) are excluded pending validation in a future cycle."
|
|
},
|
|
"schema": {
|
|
"long_column_count": 28,
|
|
"long_columns": ["fips_state", "type", "fips_county", "govid", "gov_blank", "gov_name", "county_name", "fips_state_asof", "fips_county_asof", "cog_legacy_state", "cog_legacy_county", "fips_place_code", "population", "popyear", "enrollment", "enrollyear", "function_code", "sch_level_code", "fiscal_year_end", "srvy_year", "item_code", "amt", "srv_data", "impute_flag", "is_aggregate", "canonical_govid", "harmonized_code", "survey_weight"],
|
|
"data_dictionary": "docs/data_dictionary.md"
|
|
},
|
|
"files": {
|
|
"long_partitions": [
|
|
{
|
|
"year": 2011,
|
|
"path": "data/long/year=2011/part-0.parquet",
|
|
"sha256": "7848e18497080c8980a4f89c5b386205b2c5bc90db6773827ea01ab3943d16b1",
|
|
"row_count": 496004,
|
|
"size_bytes": 2202455
|
|
},
|
|
{
|
|
"year": 2012,
|
|
"path": "data/long/year=2012/part-0.parquet",
|
|
"sha256": "b82ac82d5e35f844b26c887445601f3748438c52c998ba4e403b025941a6f170",
|
|
"row_count": 1163338,
|
|
"size_bytes": 5929917
|
|
},
|
|
{
|
|
"year": 2019,
|
|
"path": "data/long/year=2019/part-0.parquet",
|
|
"sha256": "5cbd4726dcc7d0dab5c2a05a64702e979533ae119ed0587073cd31c089e0d737",
|
|
"row_count": 318139,
|
|
"size_bytes": 1719548
|
|
},
|
|
{
|
|
"year": 2020,
|
|
"path": "data/long/year=2020/part-0.parquet",
|
|
"sha256": "ee548fec80bf1beda844fe03916ac145f10dd34c45968407cc330ec260935f00",
|
|
"row_count": 317500,
|
|
"size_bytes": 1722918
|
|
}
|
|
],
|
|
"metadata": [
|
|
{
|
|
"path": "data/canonical_alias.parquet",
|
|
"sha256": "3f617051c23a99bea322889857f7106df0c92954564afeec181df7083ee6698e",
|
|
"description": "canonical_alias.parquet"
|
|
},
|
|
{
|
|
"path": "data/canonical_fips_xwalk.parquet",
|
|
"sha256": "f98742f941269dacf8f7de5c273aa4dd4e75017a5bb70c054da35852a95a8d46",
|
|
"description": "canonical_fips_xwalk.parquet"
|
|
},
|
|
{
|
|
"path": "data/census_collection_coverage.parquet",
|
|
"sha256": "143e025616cde684da7c4442bc00d07fbd1556fabb0ea96223931b737e5d10a4",
|
|
"description": "census_collection_coverage.parquet"
|
|
},
|
|
{
|
|
"path": "data/code_set.parquet",
|
|
"sha256": "4cffcb0198dd51e4ff2b694050bb371a5f9965cdac12f25521cb628fb8e118a9",
|
|
"description": "code_set.parquet"
|
|
},
|
|
{
|
|
"path": "data/harmonization_map.parquet",
|
|
"sha256": "4cf32d0f817079ba4f28dc0ce65450d3247ebbf08d94c0c26c0d02af597bf812",
|
|
"description": "harmonization_map.parquet"
|
|
},
|
|
{
|
|
"path": "data/harmonization_recipes.parquet",
|
|
"sha256": "1133e9a0b02f8f34f5f936e55c5ecd596bb8a55d8425dcce76767f0f3203581c",
|
|
"description": "harmonization_recipes.parquet"
|
|
},
|
|
{
|
|
"path": "data/lineage_events.parquet",
|
|
"sha256": "36c16acfbe621d61010984767f1c566993b8a5f481a2c1e134c4c0a600e4502f",
|
|
"description": "lineage_events.parquet"
|
|
},
|
|
{
|
|
"path": "data/representation.parquet",
|
|
"sha256": "31ec328a7dd505a321b45f97aafff12e53d68a1a986f63509863035b22a4360d",
|
|
"description": "representation.parquet"
|
|
},
|
|
{
|
|
"path": "data/series_breaks.parquet",
|
|
"sha256": "381090660c8b8a1bee852e7f63d29b9ecaf10f71870f41c63de92017c83b6f1f",
|
|
"description": "series_breaks.parquet"
|
|
},
|
|
{
|
|
"path": "data/summary_categories.parquet",
|
|
"sha256": "e4918abf8e9e6d1372d7ddc255dc199f9c303e250f4106449241b48c68abee67",
|
|
"description": "summary_categories.parquet"
|
|
}
|
|
]
|
|
},
|
|
"series_breaks_ref": "docs/series_breaks.md",
|
|
"reader_spec_ref": "docs/reader-specification.md"
|
|
}
|