fix: rank district suggestions over the full state, not the first 500 rows

DistrictSearch built its "most arrests" list from
/estimates?state=XX&year=21-22&limit=500. That endpoint returns rows
ORDER BY LEAID, RACE, SEX at eight rows per district, so a 500-row cap is not a
sample of the state — it is the ~62 lowest-LEAID districts in it.

Measured against California (11,488 rows, 1,715 districts): the old read covered
68 districts, and 6 of the true top 8 were invisible to it. It suggested
districts with 1 and 2 arrests as the state's most notable, while San Diego
Unified (178), Fresno Unified (77) and Kern High (69) never appeared.

Replaced with a committed fixture, public/data/top_districts.json, generated by
scripts/build-top-districts.mjs. The script pages each state to completion using
meta.total from the response envelope and fails loudly on a short read, since a
silent truncation there would reintroduce exactly this bug. 51 states, 135
requests, ~104KB, following the national_rates.json precedent. Re-run it only
when a new CRDC wave lands.

The search screen also loses a multi-second fetch on every visit, and the
hardcoded "Try Derby (KS), Paterson (NJ)" hint goes with it — the real list
supersedes it. Degrades to search-only if the fixture is missing.

fetchStateDistricts() is kept for scripts and ad-hoc use, with its JSDoc now
warning that any short read ranks by LEAID.

pages.yml gains public/** in its paths filter: the fixture ships with the build,
so regenerating it has to be able to trigger a deploy on its own.
This commit is contained in:
2026-08-12 08:40:23 -04:00
parent c62d1e3068
commit 63413f9eb7
5 changed files with 5000 additions and 34 deletions
+4
View File
@@ -10,6 +10,10 @@ on:
branches: [main] branches: [main]
paths: paths:
- 'src/**' - 'src/**'
# Committed fixtures ship with the build: public/data/top_districts.json
# is what the search screen ranks its suggestions from, so regenerating it
# has to be able to trigger a deploy on its own.
- 'public/**'
- 'index.html' - 'index.html'
- 'vite.config.mjs' - 'vite.config.mjs'
- 'package.json' - 'package.json'
File diff suppressed because it is too large Load Diff
+204
View File
@@ -0,0 +1,204 @@
#!/usr/bin/env node
/**
* Builds `public/data/top_districts.json` — the "suggested districts" list the
* search screen shows before you type anything.
*
* Why this exists: the app used to build that list at runtime from
* `/estimates?state=XX&year=21-22&limit=500`. That endpoint returns rows
* `ORDER BY LEAID, RACE, SEX` at eight rows per district, so a 500-row cap is
* the ~62 *lowest-LEAID* districts in the state, not the busiest ones —
* California alone has 11,488 rows. The list was therefore ranked over a
* truncated and essentially arbitrary slice of each state.
*
* This script pages the whole state using `meta.total` from the response
* envelope, aggregates observed arrests per district, and commits the answer as
* a fixture (the `public/data/national_rates.json` precedent). The search screen
* then loses a multi-second fetch and gets a correct ranking.
*
* Read-only against the public API. Roughly 150 requests as a one-off; re-run it
* only when a new CRDC wave lands.
*
* node scripts/build-top-districts.mjs
* node scripts/build-top-districts.mjs --states NV,CA # spot-check a few
*/
import { writeFile, mkdir } from 'node:fs/promises'
import { dirname, resolve } from 'node:path'
import { fileURLToPath } from 'node:url'
const BASE_URL = process.env.CRDC_API_BASE || 'https://crdc-api.civilytics.org/api/v1'
const YEAR = '21-22'
// Pinned rather than left to the API default so a change to that default can't
// silently alter the fixture. Enrollment and observed arrests are the same in
// every specification; only the modelled columns differ, and we read none.
const MODEL = 'unified_m2_mod'
const PAGE_SIZE = 1000 // the API's LIMIT_CAP
const TOP_N = 15
const CONCURRENCY = 3
const MAX_RETRIES = 4
const ALL_STATES = [
'AL', 'AK', 'AZ', 'AR', 'CA', 'CO', 'CT', 'DE', 'DC', 'FL', 'GA', 'HI',
'ID', 'IL', 'IN', 'IA', 'KS', 'KY', 'LA', 'ME', 'MD', 'MA', 'MI', 'MN',
'MS', 'MO', 'MT', 'NE', 'NV', 'NH', 'NJ', 'NM', 'NY', 'NC', 'ND', 'OH',
'OK', 'OR', 'PA', 'RI', 'SC', 'SD', 'TN', 'TX', 'UT', 'VT', 'VA', 'WA',
'WV', 'WI', 'WY',
]
const OUT_PATH = resolve(
dirname(fileURLToPath(import.meta.url)),
'..',
'public',
'data',
'top_districts.json',
)
function parseStates() {
const flag = process.argv.indexOf('--states')
if (flag === -1) return ALL_STATES
const requested = (process.argv[flag + 1] || '').split(',').map((s) => s.trim().toUpperCase())
const unknown = requested.filter((s) => !ALL_STATES.includes(s))
if (unknown.length) throw new Error(`Unknown state code(s): ${unknown.join(', ')}`)
return requested
}
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
/** GET one page, returning the full envelope (we need `meta.total`). */
async function fetchPage(state, page) {
const params = new URLSearchParams({
state,
year: YEAR,
model: MODEL,
limit: String(PAGE_SIZE),
page: String(page),
})
const url = `${BASE_URL}/estimates?${params}`
let lastError
for (let attempt = 0; attempt <= MAX_RETRIES; attempt++) {
try {
const res = await fetch(url, { signal: AbortSignal.timeout(60000) })
if (!res.ok) throw new Error(`HTTP ${res.status} ${res.statusText}`)
const envelope = await res.json()
if (envelope.status !== 'success') throw new Error(envelope.error || 'Unknown API error')
if (!envelope.meta || typeof envelope.meta.total !== 'number') {
throw new Error('Response envelope is missing meta.total — cannot page safely')
}
return envelope
} catch (err) {
lastError = err
if (attempt === MAX_RETRIES) break
await sleep(500 * 2 ** attempt)
}
}
throw new Error(`${state} page ${page}: ${lastError.message}`)
}
async function collectState(state) {
const first = await fetchPage(state, 0)
const total = first.meta.total
const rows = [...first.data]
const pages = Math.ceil(total / PAGE_SIZE)
for (let page = 1; page < pages; page++) {
const envelope = await fetchPage(state, page)
rows.push(...envelope.data)
}
if (rows.length !== total) {
// Loud rather than silent: a short read here would quietly produce a
// truncated ranking, which is the exact bug this script exists to fix.
throw new Error(`${state}: expected ${total} rows, collected ${rows.length}`)
}
const byLeaid = new Map()
for (const row of rows) {
const leaid = row.leaid
if (!leaid) continue
const prev = byLeaid.get(leaid) || { leaid, name: row.lea_name || leaid, arrests: 0, enrollment: 0 }
byLeaid.set(leaid, {
...prev,
name: prev.name || row.lea_name || leaid,
arrests: prev.arrests + (row.observed_arrests || 0),
enrollment: prev.enrollment + (row.stu_enroll || 0),
})
}
const ranked = [...byLeaid.values()]
.filter((d) => d.arrests > 0)
.sort((a, b) => b.arrests - a.arrests || a.leaid.localeCompare(b.leaid))
.slice(0, TOP_N)
.map((d) => ({
leaid: d.leaid,
name: d.name,
arrests: d.arrests,
enrollment: d.enrollment,
rate: d.enrollment > 0 ? Math.round((d.arrests / d.enrollment) * 1000 * 100) / 100 : 0,
}))
return { state, districts: ranked, districtsSeen: byLeaid.size, rows: total, pages }
}
/** Small fixed-size worker pool — polite to a single public API host. */
async function mapWithConcurrency(items, limit, worker) {
const results = new Array(items.length)
let next = 0
const runners = Array.from({ length: Math.min(limit, items.length) }, async () => {
while (next < items.length) {
const i = next++
results[i] = await worker(items[i], i)
}
})
await Promise.all(runners)
return results
}
async function main() {
const states = parseStates()
console.log(`Fetching ${states.length} state(s) from ${BASE_URL} (year ${YEAR}, model ${MODEL})…`)
let done = 0
let requests = 0
const collected = await mapWithConcurrency(states, CONCURRENCY, async (state) => {
const result = await collectState(state)
requests += result.pages
done += 1
console.log(
` [${String(done).padStart(2)}/${states.length}] ${state}: ` +
`${result.rows} rows / ${result.pages} page(s), ` +
`${result.districtsSeen} districts, top ${result.districts.length} kept`,
)
return result
})
const byState = {}
for (const { state, districts } of collected.sort((a, b) => a.state.localeCompare(b.state))) {
byState[state] = districts
}
const payload = {
metadata: {
source: 'CRDC School Arrest Rate API (Knowles & Miller 2025)',
endpoint: `${BASE_URL}/estimates`,
year: YEAR,
model: MODEL,
description:
`Top ${TOP_N} school districts per state by total observed arrests in ${YEAR}, ` +
'summed across the eight modelled race×sex groups. Generated by ' +
'scripts/build-top-districts.mjs over the complete paged result set for each state.',
generated_states: states.length,
generated_requests: requests,
},
states: byState,
}
await mkdir(dirname(OUT_PATH), { recursive: true })
await writeFile(OUT_PATH, `${JSON.stringify(payload, null, 2)}\n`, 'utf8')
console.log(`\nWrote ${OUT_PATH} (${requests} requests, ${states.length} states).`)
}
main().catch((err) => {
console.error('\nbuild-top-districts failed:', err.message)
process.exit(1)
})
+54 -33
View File
@@ -1,10 +1,42 @@
import { useState, useEffect } from 'react' import { useState, useEffect } from 'react'
import { searchDistricts, fetchStateDistricts } from '../hooks/useApi.js' import { searchDistricts } from '../hooks/useApi.js'
const TOP_DISTRICTS_URL = `${import.meta.env.BASE_URL}data/top_districts.json`
const SUGGESTION_COUNT = 8
// Module-level cache: the fixture covers every state, so it is fetched at most
// once per page load no matter how many states the visitor browses through.
let topDistrictsPromise = null
function loadTopDistricts() {
if (!topDistrictsPromise) {
topDistrictsPromise = fetch(TOP_DISTRICTS_URL)
.then((res) => {
if (!res.ok) throw new Error(`HTTP ${res.status} loading ${TOP_DISTRICTS_URL}`)
return res.json()
})
.catch((err) => {
// Don't poison the cache with a rejected promise — let a later visit retry.
topDistrictsPromise = null
throw err
})
}
return topDistrictsPromise
}
/** /**
* District search screen with: * District search screen with:
* 1. "Interesting" suggestions — districts with the most arrests in the selected state (fetched once) * 1. "Interesting" suggestions — the districts with the most arrests in the
* selected state, read from the committed `public/data/top_districts.json`
* fixture (built by `scripts/build-top-districts.mjs`).
* 2. Live-search as you type → /api/v1/districts?q=...&state=XX * 2. Live-search as you type → /api/v1/districts?q=...&state=XX
*
* The suggestions used to be computed at runtime from
* `/estimates?state=XX&year=21-22&limit=500`. That endpoint returns rows
* `ORDER BY LEAID, RACE, SEX` at eight rows per district, so the cap selected
* the ~62 lowest-LEAID districts in the state rather than the busiest ones —
* California has 11,488 rows. The fixture is ranked over the complete result
* set, and it also removes a multi-second fetch from this screen.
*/ */
export default function DistrictSearch({ state, onSelect, onBack }) { export default function DistrictSearch({ state, onSelect, onBack }) {
const [query, setQuery] = useState('') const [query, setQuery] = useState('')
@@ -13,38 +45,32 @@ export default function DistrictSearch({ state, onSelect, onBack }) {
const [loadingSugg, setLoadingSugg] = useState(true) const [loadingSugg, setLoadingSugg] = useState(true)
const [loadingSearch, setLoadingSearch] = useState(false) const [loadingSearch, setLoadingSearch] = useState(false)
// ——— Fetch interesting suggestions (top arrests in this state) once on mount ——— // ——— Read the suggestion fixture for this state ———
useEffect(() => { useEffect(() => {
let cancelled = false let cancelled = false
async function loadSuggestions() { setLoadingSugg(true)
try {
// /estimates?state=XX&year=21-22 returns all districts; sort by observed_arrests desc client-side
const data = await fetchStateDistricts(state, '21-22', 500)
// Aggregate arrests per district (sum across race×sex groups), then sort loadTopDistricts()
const byLeaid = {} .then((fixture) => {
for (const row of data) { if (cancelled) return
if (!byLeaid[row.leaid]) { const forState = fixture?.states?.[state] || []
byLeaid[row.leaid] = { leaid: row.leaid, lea_name: row.lea_name, state: row.state, observed_arrests: 0 } setSuggestions(
} forState.slice(0, SUGGESTION_COUNT).map((d) => ({
byLeaid[row.leaid].observed_arrests += (row.observed_arrests || 0) leaid: d.leaid,
} lea_name: d.name,
state,
const sorted = Object.values(byLeaid).sort((a, b) => b.observed_arrests - a.observed_arrests) observed_arrests: d.arrests,
})),
if (!cancelled) { )
// Take top 8 for suggestions; filter to those with >0 arrests })
setSuggestions(sorted.filter(d => d.observed_arrests > 0).slice(0, 8)) .catch((err) => {
} console.error('Failed to load suggested districts:', err)
} catch (err) {
console.error('Failed to load interesting districts:', err)
if (!cancelled) setSuggestions([]) // degrade gracefully — just show search box if (!cancelled) setSuggestions([]) // degrade gracefully — just show search box
} finally { })
.finally(() => {
if (!cancelled) setLoadingSugg(false) if (!cancelled) setLoadingSugg(false)
} })
}
loadSuggestions()
return () => { cancelled = true } return () => { cancelled = true }
}, [state]) }, [state])
@@ -158,11 +184,6 @@ export default function DistrictSearch({ state, onSelect, onBack }) {
</div> </div>
)} )}
</div> </div>
{/* Hint */}
<p style={{ fontSize: '0.85rem', color: 'var(--cv-ink-3)', marginTop: 'var(--space-4)' }}>
Tip: Try districts like Derby (KS), Paterson (NJ), or Mobile County (AL) — they have notable arrest rates.
</p>
</div> </div>
) )
} }
+10 -1
View File
@@ -119,7 +119,16 @@ export async function fetchStateEstimates(state, options = {}) {
return apiFetch(`/states/${state}${qs ? `?${qs}` : ''}`) return apiFetch(`/states/${state}${qs ? `?${qs}` : ''}`)
} }
/** GET /estimates?state=XX&year=Y — all districts in a state for "interesting" suggestions */ /**
* GET /estimates?state=XX&year=Y — every estimate row in a state.
*
* Not used by the running app. The rows come back `ORDER BY LEAID, RACE, SEX`
* at eight per district, so any `limit` short of the state's full row count
* selects the lowest-LEAID districts rather than a meaningful sample —
* `DistrictSearch` reads the pre-ranked `public/data/top_districts.json`
* fixture instead. Kept for scripts and ad-hoc use; page it with `meta.total`
* as `scripts/build-top-districts.mjs` does.
*/
export async function fetchStateDistricts(state, year = '21-22', limit = 500) { export async function fetchStateDistricts(state, year = '21-22', limit = 500) {
return apiFetch(`/estimates?state=${state}&year=${year}&limit=${limit}`) return apiFetch(`/estimates?state=${state}&year=${year}&limit=${limit}`)
} }