fix: rank district suggestions over the full state, not the first 500 rows

DistrictSearch built its "most arrests" list from
/estimates?state=XX&year=21-22&limit=500. That endpoint returns rows
ORDER BY LEAID, RACE, SEX at eight rows per district, so a 500-row cap is not a
sample of the state — it is the ~62 lowest-LEAID districts in it.

Measured against California (11,488 rows, 1,715 districts): the old read covered
68 districts, and 6 of the true top 8 were invisible to it. It suggested
districts with 1 and 2 arrests as the state's most notable, while San Diego
Unified (178), Fresno Unified (77) and Kern High (69) never appeared.

Replaced with a committed fixture, public/data/top_districts.json, generated by
scripts/build-top-districts.mjs. The script pages each state to completion using
meta.total from the response envelope and fails loudly on a short read, since a
silent truncation there would reintroduce exactly this bug. 51 states, 135
requests, ~104KB, following the national_rates.json precedent. Re-run it only
when a new CRDC wave lands.

The search screen also loses a multi-second fetch on every visit, and the
hardcoded "Try Derby (KS), Paterson (NJ)" hint goes with it — the real list
supersedes it. Degrades to search-only if the fixture is missing.

fetchStateDistricts() is kept for scripts and ad-hoc use, with its JSDoc now
warning that any short read ranks by LEAID.

pages.yml gains public/** in its paths filter: the fixture ships with the build,
so regenerating it has to be able to trigger a deploy on its own.
This commit is contained in:
2026-08-12 08:40:23 -04:00
parent c62d1e3068
commit 63413f9eb7
5 changed files with 5000 additions and 34 deletions
+204
View File
@@ -0,0 +1,204 @@
#!/usr/bin/env node
/**
* Builds `public/data/top_districts.json` — the "suggested districts" list the
* search screen shows before you type anything.
*
* Why this exists: the app used to build that list at runtime from
* `/estimates?state=XX&year=21-22&limit=500`. That endpoint returns rows
* `ORDER BY LEAID, RACE, SEX` at eight rows per district, so a 500-row cap is
* the ~62 *lowest-LEAID* districts in the state, not the busiest ones —
* California alone has 11,488 rows. The list was therefore ranked over a
* truncated and essentially arbitrary slice of each state.
*
* This script pages the whole state using `meta.total` from the response
* envelope, aggregates observed arrests per district, and commits the answer as
* a fixture (the `public/data/national_rates.json` precedent). The search screen
* then loses a multi-second fetch and gets a correct ranking.
*
* Read-only against the public API. Roughly 150 requests as a one-off; re-run it
* only when a new CRDC wave lands.
*
* node scripts/build-top-districts.mjs
* node scripts/build-top-districts.mjs --states NV,CA # spot-check a few
*/
import { writeFile, mkdir } from 'node:fs/promises'
import { dirname, resolve } from 'node:path'
import { fileURLToPath } from 'node:url'
const BASE_URL = process.env.CRDC_API_BASE || 'https://crdc-api.civilytics.org/api/v1'
const YEAR = '21-22'
// Pinned rather than left to the API default so a change to that default can't
// silently alter the fixture. Enrollment and observed arrests are the same in
// every specification; only the modelled columns differ, and we read none.
const MODEL = 'unified_m2_mod'
const PAGE_SIZE = 1000 // the API's LIMIT_CAP
const TOP_N = 15
const CONCURRENCY = 3
const MAX_RETRIES = 4
const ALL_STATES = [
'AL', 'AK', 'AZ', 'AR', 'CA', 'CO', 'CT', 'DE', 'DC', 'FL', 'GA', 'HI',
'ID', 'IL', 'IN', 'IA', 'KS', 'KY', 'LA', 'ME', 'MD', 'MA', 'MI', 'MN',
'MS', 'MO', 'MT', 'NE', 'NV', 'NH', 'NJ', 'NM', 'NY', 'NC', 'ND', 'OH',
'OK', 'OR', 'PA', 'RI', 'SC', 'SD', 'TN', 'TX', 'UT', 'VT', 'VA', 'WA',
'WV', 'WI', 'WY',
]
const OUT_PATH = resolve(
dirname(fileURLToPath(import.meta.url)),
'..',
'public',
'data',
'top_districts.json',
)
function parseStates() {
const flag = process.argv.indexOf('--states')
if (flag === -1) return ALL_STATES
const requested = (process.argv[flag + 1] || '').split(',').map((s) => s.trim().toUpperCase())
const unknown = requested.filter((s) => !ALL_STATES.includes(s))
if (unknown.length) throw new Error(`Unknown state code(s): ${unknown.join(', ')}`)
return requested
}
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
/** GET one page, returning the full envelope (we need `meta.total`). */
async function fetchPage(state, page) {
const params = new URLSearchParams({
state,
year: YEAR,
model: MODEL,
limit: String(PAGE_SIZE),
page: String(page),
})
const url = `${BASE_URL}/estimates?${params}`
let lastError
for (let attempt = 0; attempt <= MAX_RETRIES; attempt++) {
try {
const res = await fetch(url, { signal: AbortSignal.timeout(60000) })
if (!res.ok) throw new Error(`HTTP ${res.status} ${res.statusText}`)
const envelope = await res.json()
if (envelope.status !== 'success') throw new Error(envelope.error || 'Unknown API error')
if (!envelope.meta || typeof envelope.meta.total !== 'number') {
throw new Error('Response envelope is missing meta.total — cannot page safely')
}
return envelope
} catch (err) {
lastError = err
if (attempt === MAX_RETRIES) break
await sleep(500 * 2 ** attempt)
}
}
throw new Error(`${state} page ${page}: ${lastError.message}`)
}
async function collectState(state) {
const first = await fetchPage(state, 0)
const total = first.meta.total
const rows = [...first.data]
const pages = Math.ceil(total / PAGE_SIZE)
for (let page = 1; page < pages; page++) {
const envelope = await fetchPage(state, page)
rows.push(...envelope.data)
}
if (rows.length !== total) {
// Loud rather than silent: a short read here would quietly produce a
// truncated ranking, which is the exact bug this script exists to fix.
throw new Error(`${state}: expected ${total} rows, collected ${rows.length}`)
}
const byLeaid = new Map()
for (const row of rows) {
const leaid = row.leaid
if (!leaid) continue
const prev = byLeaid.get(leaid) || { leaid, name: row.lea_name || leaid, arrests: 0, enrollment: 0 }
byLeaid.set(leaid, {
...prev,
name: prev.name || row.lea_name || leaid,
arrests: prev.arrests + (row.observed_arrests || 0),
enrollment: prev.enrollment + (row.stu_enroll || 0),
})
}
const ranked = [...byLeaid.values()]
.filter((d) => d.arrests > 0)
.sort((a, b) => b.arrests - a.arrests || a.leaid.localeCompare(b.leaid))
.slice(0, TOP_N)
.map((d) => ({
leaid: d.leaid,
name: d.name,
arrests: d.arrests,
enrollment: d.enrollment,
rate: d.enrollment > 0 ? Math.round((d.arrests / d.enrollment) * 1000 * 100) / 100 : 0,
}))
return { state, districts: ranked, districtsSeen: byLeaid.size, rows: total, pages }
}
/** Small fixed-size worker pool — polite to a single public API host. */
async function mapWithConcurrency(items, limit, worker) {
const results = new Array(items.length)
let next = 0
const runners = Array.from({ length: Math.min(limit, items.length) }, async () => {
while (next < items.length) {
const i = next++
results[i] = await worker(items[i], i)
}
})
await Promise.all(runners)
return results
}
async function main() {
const states = parseStates()
console.log(`Fetching ${states.length} state(s) from ${BASE_URL} (year ${YEAR}, model ${MODEL})…`)
let done = 0
let requests = 0
const collected = await mapWithConcurrency(states, CONCURRENCY, async (state) => {
const result = await collectState(state)
requests += result.pages
done += 1
console.log(
` [${String(done).padStart(2)}/${states.length}] ${state}: ` +
`${result.rows} rows / ${result.pages} page(s), ` +
`${result.districtsSeen} districts, top ${result.districts.length} kept`,
)
return result
})
const byState = {}
for (const { state, districts } of collected.sort((a, b) => a.state.localeCompare(b.state))) {
byState[state] = districts
}
const payload = {
metadata: {
source: 'CRDC School Arrest Rate API (Knowles & Miller 2025)',
endpoint: `${BASE_URL}/estimates`,
year: YEAR,
model: MODEL,
description:
`Top ${TOP_N} school districts per state by total observed arrests in ${YEAR}, ` +
'summed across the eight modelled race×sex groups. Generated by ' +
'scripts/build-top-districts.mjs over the complete paged result set for each state.',
generated_states: states.length,
generated_requests: requests,
},
states: byState,
}
await mkdir(dirname(OUT_PATH), { recursive: true })
await writeFile(OUT_PATH, `${JSON.stringify(payload, null, 2)}\n`, 'utf8')
console.log(`\nWrote ${OUT_PATH} (${requests} requests, ${states.length} states).`)
}
main().catch((err) => {
console.error('\nbuild-top-districts failed:', err.message)
process.exit(1)
})