fix: rank district suggestions over the full state, not the first 500 rows
DistrictSearch built its "most arrests" list from /estimates?state=XX&year=21-22&limit=500. That endpoint returns rows ORDER BY LEAID, RACE, SEX at eight rows per district, so a 500-row cap is not a sample of the state — it is the ~62 lowest-LEAID districts in it. Measured against California (11,488 rows, 1,715 districts): the old read covered 68 districts, and 6 of the true top 8 were invisible to it. It suggested districts with 1 and 2 arrests as the state's most notable, while San Diego Unified (178), Fresno Unified (77) and Kern High (69) never appeared. Replaced with a committed fixture, public/data/top_districts.json, generated by scripts/build-top-districts.mjs. The script pages each state to completion using meta.total from the response envelope and fails loudly on a short read, since a silent truncation there would reintroduce exactly this bug. 51 states, 135 requests, ~104KB, following the national_rates.json precedent. Re-run it only when a new CRDC wave lands. The search screen also loses a multi-second fetch on every visit, and the hardcoded "Try Derby (KS), Paterson (NJ)" hint goes with it — the real list supersedes it. Degrades to search-only if the fixture is missing. fetchStateDistricts() is kept for scripts and ad-hoc use, with its JSDoc now warning that any short read ranks by LEAID. pages.yml gains public/** in its paths filter: the fixture ships with the build, so regenerating it has to be able to trigger a deploy on its own.
This commit is contained in:
@@ -0,0 +1,204 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* Builds `public/data/top_districts.json` — the "suggested districts" list the
|
||||
* search screen shows before you type anything.
|
||||
*
|
||||
* Why this exists: the app used to build that list at runtime from
|
||||
* `/estimates?state=XX&year=21-22&limit=500`. That endpoint returns rows
|
||||
* `ORDER BY LEAID, RACE, SEX` at eight rows per district, so a 500-row cap is
|
||||
* the ~62 *lowest-LEAID* districts in the state, not the busiest ones —
|
||||
* California alone has 11,488 rows. The list was therefore ranked over a
|
||||
* truncated and essentially arbitrary slice of each state.
|
||||
*
|
||||
* This script pages the whole state using `meta.total` from the response
|
||||
* envelope, aggregates observed arrests per district, and commits the answer as
|
||||
* a fixture (the `public/data/national_rates.json` precedent). The search screen
|
||||
* then loses a multi-second fetch and gets a correct ranking.
|
||||
*
|
||||
* Read-only against the public API. Roughly 150 requests as a one-off; re-run it
|
||||
* only when a new CRDC wave lands.
|
||||
*
|
||||
* node scripts/build-top-districts.mjs
|
||||
* node scripts/build-top-districts.mjs --states NV,CA # spot-check a few
|
||||
*/
|
||||
|
||||
import { writeFile, mkdir } from 'node:fs/promises'
|
||||
import { dirname, resolve } from 'node:path'
|
||||
import { fileURLToPath } from 'node:url'
|
||||
|
||||
const BASE_URL = process.env.CRDC_API_BASE || 'https://crdc-api.civilytics.org/api/v1'
|
||||
const YEAR = '21-22'
|
||||
// Pinned rather than left to the API default so a change to that default can't
|
||||
// silently alter the fixture. Enrollment and observed arrests are the same in
|
||||
// every specification; only the modelled columns differ, and we read none.
|
||||
const MODEL = 'unified_m2_mod'
|
||||
const PAGE_SIZE = 1000 // the API's LIMIT_CAP
|
||||
const TOP_N = 15
|
||||
const CONCURRENCY = 3
|
||||
const MAX_RETRIES = 4
|
||||
|
||||
const ALL_STATES = [
|
||||
'AL', 'AK', 'AZ', 'AR', 'CA', 'CO', 'CT', 'DE', 'DC', 'FL', 'GA', 'HI',
|
||||
'ID', 'IL', 'IN', 'IA', 'KS', 'KY', 'LA', 'ME', 'MD', 'MA', 'MI', 'MN',
|
||||
'MS', 'MO', 'MT', 'NE', 'NV', 'NH', 'NJ', 'NM', 'NY', 'NC', 'ND', 'OH',
|
||||
'OK', 'OR', 'PA', 'RI', 'SC', 'SD', 'TN', 'TX', 'UT', 'VT', 'VA', 'WA',
|
||||
'WV', 'WI', 'WY',
|
||||
]
|
||||
|
||||
const OUT_PATH = resolve(
|
||||
dirname(fileURLToPath(import.meta.url)),
|
||||
'..',
|
||||
'public',
|
||||
'data',
|
||||
'top_districts.json',
|
||||
)
|
||||
|
||||
function parseStates() {
|
||||
const flag = process.argv.indexOf('--states')
|
||||
if (flag === -1) return ALL_STATES
|
||||
const requested = (process.argv[flag + 1] || '').split(',').map((s) => s.trim().toUpperCase())
|
||||
const unknown = requested.filter((s) => !ALL_STATES.includes(s))
|
||||
if (unknown.length) throw new Error(`Unknown state code(s): ${unknown.join(', ')}`)
|
||||
return requested
|
||||
}
|
||||
|
||||
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
|
||||
|
||||
/** GET one page, returning the full envelope (we need `meta.total`). */
|
||||
async function fetchPage(state, page) {
|
||||
const params = new URLSearchParams({
|
||||
state,
|
||||
year: YEAR,
|
||||
model: MODEL,
|
||||
limit: String(PAGE_SIZE),
|
||||
page: String(page),
|
||||
})
|
||||
const url = `${BASE_URL}/estimates?${params}`
|
||||
|
||||
let lastError
|
||||
for (let attempt = 0; attempt <= MAX_RETRIES; attempt++) {
|
||||
try {
|
||||
const res = await fetch(url, { signal: AbortSignal.timeout(60000) })
|
||||
if (!res.ok) throw new Error(`HTTP ${res.status} ${res.statusText}`)
|
||||
const envelope = await res.json()
|
||||
if (envelope.status !== 'success') throw new Error(envelope.error || 'Unknown API error')
|
||||
if (!envelope.meta || typeof envelope.meta.total !== 'number') {
|
||||
throw new Error('Response envelope is missing meta.total — cannot page safely')
|
||||
}
|
||||
return envelope
|
||||
} catch (err) {
|
||||
lastError = err
|
||||
if (attempt === MAX_RETRIES) break
|
||||
await sleep(500 * 2 ** attempt)
|
||||
}
|
||||
}
|
||||
throw new Error(`${state} page ${page}: ${lastError.message}`)
|
||||
}
|
||||
|
||||
async function collectState(state) {
|
||||
const first = await fetchPage(state, 0)
|
||||
const total = first.meta.total
|
||||
const rows = [...first.data]
|
||||
|
||||
const pages = Math.ceil(total / PAGE_SIZE)
|
||||
for (let page = 1; page < pages; page++) {
|
||||
const envelope = await fetchPage(state, page)
|
||||
rows.push(...envelope.data)
|
||||
}
|
||||
|
||||
if (rows.length !== total) {
|
||||
// Loud rather than silent: a short read here would quietly produce a
|
||||
// truncated ranking, which is the exact bug this script exists to fix.
|
||||
throw new Error(`${state}: expected ${total} rows, collected ${rows.length}`)
|
||||
}
|
||||
|
||||
const byLeaid = new Map()
|
||||
for (const row of rows) {
|
||||
const leaid = row.leaid
|
||||
if (!leaid) continue
|
||||
const prev = byLeaid.get(leaid) || { leaid, name: row.lea_name || leaid, arrests: 0, enrollment: 0 }
|
||||
byLeaid.set(leaid, {
|
||||
...prev,
|
||||
name: prev.name || row.lea_name || leaid,
|
||||
arrests: prev.arrests + (row.observed_arrests || 0),
|
||||
enrollment: prev.enrollment + (row.stu_enroll || 0),
|
||||
})
|
||||
}
|
||||
|
||||
const ranked = [...byLeaid.values()]
|
||||
.filter((d) => d.arrests > 0)
|
||||
.sort((a, b) => b.arrests - a.arrests || a.leaid.localeCompare(b.leaid))
|
||||
.slice(0, TOP_N)
|
||||
.map((d) => ({
|
||||
leaid: d.leaid,
|
||||
name: d.name,
|
||||
arrests: d.arrests,
|
||||
enrollment: d.enrollment,
|
||||
rate: d.enrollment > 0 ? Math.round((d.arrests / d.enrollment) * 1000 * 100) / 100 : 0,
|
||||
}))
|
||||
|
||||
return { state, districts: ranked, districtsSeen: byLeaid.size, rows: total, pages }
|
||||
}
|
||||
|
||||
/** Small fixed-size worker pool — polite to a single public API host. */
|
||||
async function mapWithConcurrency(items, limit, worker) {
|
||||
const results = new Array(items.length)
|
||||
let next = 0
|
||||
const runners = Array.from({ length: Math.min(limit, items.length) }, async () => {
|
||||
while (next < items.length) {
|
||||
const i = next++
|
||||
results[i] = await worker(items[i], i)
|
||||
}
|
||||
})
|
||||
await Promise.all(runners)
|
||||
return results
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const states = parseStates()
|
||||
console.log(`Fetching ${states.length} state(s) from ${BASE_URL} (year ${YEAR}, model ${MODEL})…`)
|
||||
|
||||
let done = 0
|
||||
let requests = 0
|
||||
const collected = await mapWithConcurrency(states, CONCURRENCY, async (state) => {
|
||||
const result = await collectState(state)
|
||||
requests += result.pages
|
||||
done += 1
|
||||
console.log(
|
||||
` [${String(done).padStart(2)}/${states.length}] ${state}: ` +
|
||||
`${result.rows} rows / ${result.pages} page(s), ` +
|
||||
`${result.districtsSeen} districts, top ${result.districts.length} kept`,
|
||||
)
|
||||
return result
|
||||
})
|
||||
|
||||
const byState = {}
|
||||
for (const { state, districts } of collected.sort((a, b) => a.state.localeCompare(b.state))) {
|
||||
byState[state] = districts
|
||||
}
|
||||
|
||||
const payload = {
|
||||
metadata: {
|
||||
source: 'CRDC School Arrest Rate API (Knowles & Miller 2025)',
|
||||
endpoint: `${BASE_URL}/estimates`,
|
||||
year: YEAR,
|
||||
model: MODEL,
|
||||
description:
|
||||
`Top ${TOP_N} school districts per state by total observed arrests in ${YEAR}, ` +
|
||||
'summed across the eight modelled race×sex groups. Generated by ' +
|
||||
'scripts/build-top-districts.mjs over the complete paged result set for each state.',
|
||||
generated_states: states.length,
|
||||
generated_requests: requests,
|
||||
},
|
||||
states: byState,
|
||||
}
|
||||
|
||||
await mkdir(dirname(OUT_PATH), { recursive: true })
|
||||
await writeFile(OUT_PATH, `${JSON.stringify(payload, null, 2)}\n`, 'utf8')
|
||||
console.log(`\nWrote ${OUT_PATH} (${requests} requests, ${states.length} states).`)
|
||||
}
|
||||
|
||||
main().catch((err) => {
|
||||
console.error('\nbuild-top-districts failed:', err.message)
|
||||
process.exit(1)
|
||||
})
|
||||
Reference in New Issue
Block a user