Files
crdc-demo/scripts/build-top-districts.mjs
jared 63413f9eb7 fix: rank district suggestions over the full state, not the first 500 rows
DistrictSearch built its "most arrests" list from
/estimates?state=XX&year=21-22&limit=500. That endpoint returns rows
ORDER BY LEAID, RACE, SEX at eight rows per district, so a 500-row cap is not a
sample of the state — it is the ~62 lowest-LEAID districts in it.

Measured against California (11,488 rows, 1,715 districts): the old read covered
68 districts, and 6 of the true top 8 were invisible to it. It suggested
districts with 1 and 2 arrests as the state's most notable, while San Diego
Unified (178), Fresno Unified (77) and Kern High (69) never appeared.

Replaced with a committed fixture, public/data/top_districts.json, generated by
scripts/build-top-districts.mjs. The script pages each state to completion using
meta.total from the response envelope and fails loudly on a short read, since a
silent truncation there would reintroduce exactly this bug. 51 states, 135
requests, ~104KB, following the national_rates.json precedent. Re-run it only
when a new CRDC wave lands.

The search screen also loses a multi-second fetch on every visit, and the
hardcoded "Try Derby (KS), Paterson (NJ)" hint goes with it — the real list
supersedes it. Degrades to search-only if the fixture is missing.

fetchStateDistricts() is kept for scripts and ad-hoc use, with its JSDoc now
warning that any short read ranks by LEAID.

pages.yml gains public/** in its paths filter: the fixture ships with the build,
so regenerating it has to be able to trigger a deploy on its own.
2026-08-12 08:40:23 -04:00

205 lines
7.2 KiB
JavaScript
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env node
/**
* Builds `public/data/top_districts.json` — the "suggested districts" list the
* search screen shows before you type anything.
*
* Why this exists: the app used to build that list at runtime from
* `/estimates?state=XX&year=21-22&limit=500`. That endpoint returns rows
* `ORDER BY LEAID, RACE, SEX` at eight rows per district, so a 500-row cap is
* the ~62 *lowest-LEAID* districts in the state, not the busiest ones —
* California alone has 11,488 rows. The list was therefore ranked over a
* truncated and essentially arbitrary slice of each state.
*
* This script pages the whole state using `meta.total` from the response
* envelope, aggregates observed arrests per district, and commits the answer as
* a fixture (the `public/data/national_rates.json` precedent). The search screen
* then loses a multi-second fetch and gets a correct ranking.
*
* Read-only against the public API. Roughly 150 requests as a one-off; re-run it
* only when a new CRDC wave lands.
*
* node scripts/build-top-districts.mjs
* node scripts/build-top-districts.mjs --states NV,CA # spot-check a few
*/
import { writeFile, mkdir } from 'node:fs/promises'
import { dirname, resolve } from 'node:path'
import { fileURLToPath } from 'node:url'
const BASE_URL = process.env.CRDC_API_BASE || 'https://crdc-api.civilytics.org/api/v1'
const YEAR = '21-22'
// Pinned rather than left to the API default so a change to that default can't
// silently alter the fixture. Enrollment and observed arrests are the same in
// every specification; only the modelled columns differ, and we read none.
const MODEL = 'unified_m2_mod'
const PAGE_SIZE = 1000 // the API's LIMIT_CAP
const TOP_N = 15
const CONCURRENCY = 3
const MAX_RETRIES = 4
const ALL_STATES = [
'AL', 'AK', 'AZ', 'AR', 'CA', 'CO', 'CT', 'DE', 'DC', 'FL', 'GA', 'HI',
'ID', 'IL', 'IN', 'IA', 'KS', 'KY', 'LA', 'ME', 'MD', 'MA', 'MI', 'MN',
'MS', 'MO', 'MT', 'NE', 'NV', 'NH', 'NJ', 'NM', 'NY', 'NC', 'ND', 'OH',
'OK', 'OR', 'PA', 'RI', 'SC', 'SD', 'TN', 'TX', 'UT', 'VT', 'VA', 'WA',
'WV', 'WI', 'WY',
]
const OUT_PATH = resolve(
dirname(fileURLToPath(import.meta.url)),
'..',
'public',
'data',
'top_districts.json',
)
function parseStates() {
const flag = process.argv.indexOf('--states')
if (flag === -1) return ALL_STATES
const requested = (process.argv[flag + 1] || '').split(',').map((s) => s.trim().toUpperCase())
const unknown = requested.filter((s) => !ALL_STATES.includes(s))
if (unknown.length) throw new Error(`Unknown state code(s): ${unknown.join(', ')}`)
return requested
}
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
/** GET one page, returning the full envelope (we need `meta.total`). */
async function fetchPage(state, page) {
const params = new URLSearchParams({
state,
year: YEAR,
model: MODEL,
limit: String(PAGE_SIZE),
page: String(page),
})
const url = `${BASE_URL}/estimates?${params}`
let lastError
for (let attempt = 0; attempt <= MAX_RETRIES; attempt++) {
try {
const res = await fetch(url, { signal: AbortSignal.timeout(60000) })
if (!res.ok) throw new Error(`HTTP ${res.status} ${res.statusText}`)
const envelope = await res.json()
if (envelope.status !== 'success') throw new Error(envelope.error || 'Unknown API error')
if (!envelope.meta || typeof envelope.meta.total !== 'number') {
throw new Error('Response envelope is missing meta.total — cannot page safely')
}
return envelope
} catch (err) {
lastError = err
if (attempt === MAX_RETRIES) break
await sleep(500 * 2 ** attempt)
}
}
throw new Error(`${state} page ${page}: ${lastError.message}`)
}
async function collectState(state) {
const first = await fetchPage(state, 0)
const total = first.meta.total
const rows = [...first.data]
const pages = Math.ceil(total / PAGE_SIZE)
for (let page = 1; page < pages; page++) {
const envelope = await fetchPage(state, page)
rows.push(...envelope.data)
}
if (rows.length !== total) {
// Loud rather than silent: a short read here would quietly produce a
// truncated ranking, which is the exact bug this script exists to fix.
throw new Error(`${state}: expected ${total} rows, collected ${rows.length}`)
}
const byLeaid = new Map()
for (const row of rows) {
const leaid = row.leaid
if (!leaid) continue
const prev = byLeaid.get(leaid) || { leaid, name: row.lea_name || leaid, arrests: 0, enrollment: 0 }
byLeaid.set(leaid, {
...prev,
name: prev.name || row.lea_name || leaid,
arrests: prev.arrests + (row.observed_arrests || 0),
enrollment: prev.enrollment + (row.stu_enroll || 0),
})
}
const ranked = [...byLeaid.values()]
.filter((d) => d.arrests > 0)
.sort((a, b) => b.arrests - a.arrests || a.leaid.localeCompare(b.leaid))
.slice(0, TOP_N)
.map((d) => ({
leaid: d.leaid,
name: d.name,
arrests: d.arrests,
enrollment: d.enrollment,
rate: d.enrollment > 0 ? Math.round((d.arrests / d.enrollment) * 1000 * 100) / 100 : 0,
}))
return { state, districts: ranked, districtsSeen: byLeaid.size, rows: total, pages }
}
/** Small fixed-size worker pool — polite to a single public API host. */
async function mapWithConcurrency(items, limit, worker) {
const results = new Array(items.length)
let next = 0
const runners = Array.from({ length: Math.min(limit, items.length) }, async () => {
while (next < items.length) {
const i = next++
results[i] = await worker(items[i], i)
}
})
await Promise.all(runners)
return results
}
async function main() {
const states = parseStates()
console.log(`Fetching ${states.length} state(s) from ${BASE_URL} (year ${YEAR}, model ${MODEL})…`)
let done = 0
let requests = 0
const collected = await mapWithConcurrency(states, CONCURRENCY, async (state) => {
const result = await collectState(state)
requests += result.pages
done += 1
console.log(
` [${String(done).padStart(2)}/${states.length}] ${state}: ` +
`${result.rows} rows / ${result.pages} page(s), ` +
`${result.districtsSeen} districts, top ${result.districts.length} kept`,
)
return result
})
const byState = {}
for (const { state, districts } of collected.sort((a, b) => a.state.localeCompare(b.state))) {
byState[state] = districts
}
const payload = {
metadata: {
source: 'CRDC School Arrest Rate API (Knowles & Miller 2025)',
endpoint: `${BASE_URL}/estimates`,
year: YEAR,
model: MODEL,
description:
`Top ${TOP_N} school districts per state by total observed arrests in ${YEAR}, ` +
'summed across the eight modelled race×sex groups. Generated by ' +
'scripts/build-top-districts.mjs over the complete paged result set for each state.',
generated_states: states.length,
generated_requests: requests,
},
states: byState,
}
await mkdir(dirname(OUT_PATH), { recursive: true })
await writeFile(OUT_PATH, `${JSON.stringify(payload, null, 2)}\n`, 'utf8')
console.log(`\nWrote ${OUT_PATH} (${requests} requests, ${states.length} states).`)
}
main().catch((err) => {
console.error('\nbuild-top-districts failed:', err.message)
process.exit(1)
})