fix: rank district suggestions over the full state, not the first 500 rows
DistrictSearch built its "most arrests" list from /estimates?state=XX&year=21-22&limit=500. That endpoint returns rows ORDER BY LEAID, RACE, SEX at eight rows per district, so a 500-row cap is not a sample of the state — it is the ~62 lowest-LEAID districts in it. Measured against California (11,488 rows, 1,715 districts): the old read covered 68 districts, and 6 of the true top 8 were invisible to it. It suggested districts with 1 and 2 arrests as the state's most notable, while San Diego Unified (178), Fresno Unified (77) and Kern High (69) never appeared. Replaced with a committed fixture, public/data/top_districts.json, generated by scripts/build-top-districts.mjs. The script pages each state to completion using meta.total from the response envelope and fails loudly on a short read, since a silent truncation there would reintroduce exactly this bug. 51 states, 135 requests, ~104KB, following the national_rates.json precedent. Re-run it only when a new CRDC wave lands. The search screen also loses a multi-second fetch on every visit, and the hardcoded "Try Derby (KS), Paterson (NJ)" hint goes with it — the real list supersedes it. Degrades to search-only if the fixture is missing. fetchStateDistricts() is kept for scripts and ad-hoc use, with its JSDoc now warning that any short read ranks by LEAID. pages.yml gains public/** in its paths filter: the fixture ships with the build, so regenerating it has to be able to trigger a deploy on its own.
This commit is contained in:
@@ -10,6 +10,10 @@ on:
|
|||||||
branches: [main]
|
branches: [main]
|
||||||
paths:
|
paths:
|
||||||
- 'src/**'
|
- 'src/**'
|
||||||
|
# Committed fixtures ship with the build: public/data/top_districts.json
|
||||||
|
# is what the search screen ranks its suggestions from, so regenerating it
|
||||||
|
# has to be able to trigger a deploy on its own.
|
||||||
|
- 'public/**'
|
||||||
- 'index.html'
|
- 'index.html'
|
||||||
- 'vite.config.mjs'
|
- 'vite.config.mjs'
|
||||||
- 'package.json'
|
- 'package.json'
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,204 @@
|
|||||||
|
#!/usr/bin/env node
|
||||||
|
/**
|
||||||
|
* Builds `public/data/top_districts.json` — the "suggested districts" list the
|
||||||
|
* search screen shows before you type anything.
|
||||||
|
*
|
||||||
|
* Why this exists: the app used to build that list at runtime from
|
||||||
|
* `/estimates?state=XX&year=21-22&limit=500`. That endpoint returns rows
|
||||||
|
* `ORDER BY LEAID, RACE, SEX` at eight rows per district, so a 500-row cap is
|
||||||
|
* the ~62 *lowest-LEAID* districts in the state, not the busiest ones —
|
||||||
|
* California alone has 11,488 rows. The list was therefore ranked over a
|
||||||
|
* truncated and essentially arbitrary slice of each state.
|
||||||
|
*
|
||||||
|
* This script pages the whole state using `meta.total` from the response
|
||||||
|
* envelope, aggregates observed arrests per district, and commits the answer as
|
||||||
|
* a fixture (the `public/data/national_rates.json` precedent). The search screen
|
||||||
|
* then loses a multi-second fetch and gets a correct ranking.
|
||||||
|
*
|
||||||
|
* Read-only against the public API. Roughly 150 requests as a one-off; re-run it
|
||||||
|
* only when a new CRDC wave lands.
|
||||||
|
*
|
||||||
|
* node scripts/build-top-districts.mjs
|
||||||
|
* node scripts/build-top-districts.mjs --states NV,CA # spot-check a few
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { writeFile, mkdir } from 'node:fs/promises'
|
||||||
|
import { dirname, resolve } from 'node:path'
|
||||||
|
import { fileURLToPath } from 'node:url'
|
||||||
|
|
||||||
|
const BASE_URL = process.env.CRDC_API_BASE || 'https://crdc-api.civilytics.org/api/v1'
|
||||||
|
const YEAR = '21-22'
|
||||||
|
// Pinned rather than left to the API default so a change to that default can't
|
||||||
|
// silently alter the fixture. Enrollment and observed arrests are the same in
|
||||||
|
// every specification; only the modelled columns differ, and we read none.
|
||||||
|
const MODEL = 'unified_m2_mod'
|
||||||
|
const PAGE_SIZE = 1000 // the API's LIMIT_CAP
|
||||||
|
const TOP_N = 15
|
||||||
|
const CONCURRENCY = 3
|
||||||
|
const MAX_RETRIES = 4
|
||||||
|
|
||||||
|
const ALL_STATES = [
|
||||||
|
'AL', 'AK', 'AZ', 'AR', 'CA', 'CO', 'CT', 'DE', 'DC', 'FL', 'GA', 'HI',
|
||||||
|
'ID', 'IL', 'IN', 'IA', 'KS', 'KY', 'LA', 'ME', 'MD', 'MA', 'MI', 'MN',
|
||||||
|
'MS', 'MO', 'MT', 'NE', 'NV', 'NH', 'NJ', 'NM', 'NY', 'NC', 'ND', 'OH',
|
||||||
|
'OK', 'OR', 'PA', 'RI', 'SC', 'SD', 'TN', 'TX', 'UT', 'VT', 'VA', 'WA',
|
||||||
|
'WV', 'WI', 'WY',
|
||||||
|
]
|
||||||
|
|
||||||
|
const OUT_PATH = resolve(
|
||||||
|
dirname(fileURLToPath(import.meta.url)),
|
||||||
|
'..',
|
||||||
|
'public',
|
||||||
|
'data',
|
||||||
|
'top_districts.json',
|
||||||
|
)
|
||||||
|
|
||||||
|
function parseStates() {
|
||||||
|
const flag = process.argv.indexOf('--states')
|
||||||
|
if (flag === -1) return ALL_STATES
|
||||||
|
const requested = (process.argv[flag + 1] || '').split(',').map((s) => s.trim().toUpperCase())
|
||||||
|
const unknown = requested.filter((s) => !ALL_STATES.includes(s))
|
||||||
|
if (unknown.length) throw new Error(`Unknown state code(s): ${unknown.join(', ')}`)
|
||||||
|
return requested
|
||||||
|
}
|
||||||
|
|
||||||
|
const sleep = (ms) => new Promise((r) => setTimeout(r, ms))
|
||||||
|
|
||||||
|
/** GET one page, returning the full envelope (we need `meta.total`). */
|
||||||
|
async function fetchPage(state, page) {
|
||||||
|
const params = new URLSearchParams({
|
||||||
|
state,
|
||||||
|
year: YEAR,
|
||||||
|
model: MODEL,
|
||||||
|
limit: String(PAGE_SIZE),
|
||||||
|
page: String(page),
|
||||||
|
})
|
||||||
|
const url = `${BASE_URL}/estimates?${params}`
|
||||||
|
|
||||||
|
let lastError
|
||||||
|
for (let attempt = 0; attempt <= MAX_RETRIES; attempt++) {
|
||||||
|
try {
|
||||||
|
const res = await fetch(url, { signal: AbortSignal.timeout(60000) })
|
||||||
|
if (!res.ok) throw new Error(`HTTP ${res.status} ${res.statusText}`)
|
||||||
|
const envelope = await res.json()
|
||||||
|
if (envelope.status !== 'success') throw new Error(envelope.error || 'Unknown API error')
|
||||||
|
if (!envelope.meta || typeof envelope.meta.total !== 'number') {
|
||||||
|
throw new Error('Response envelope is missing meta.total — cannot page safely')
|
||||||
|
}
|
||||||
|
return envelope
|
||||||
|
} catch (err) {
|
||||||
|
lastError = err
|
||||||
|
if (attempt === MAX_RETRIES) break
|
||||||
|
await sleep(500 * 2 ** attempt)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
throw new Error(`${state} page ${page}: ${lastError.message}`)
|
||||||
|
}
|
||||||
|
|
||||||
|
async function collectState(state) {
|
||||||
|
const first = await fetchPage(state, 0)
|
||||||
|
const total = first.meta.total
|
||||||
|
const rows = [...first.data]
|
||||||
|
|
||||||
|
const pages = Math.ceil(total / PAGE_SIZE)
|
||||||
|
for (let page = 1; page < pages; page++) {
|
||||||
|
const envelope = await fetchPage(state, page)
|
||||||
|
rows.push(...envelope.data)
|
||||||
|
}
|
||||||
|
|
||||||
|
if (rows.length !== total) {
|
||||||
|
// Loud rather than silent: a short read here would quietly produce a
|
||||||
|
// truncated ranking, which is the exact bug this script exists to fix.
|
||||||
|
throw new Error(`${state}: expected ${total} rows, collected ${rows.length}`)
|
||||||
|
}
|
||||||
|
|
||||||
|
const byLeaid = new Map()
|
||||||
|
for (const row of rows) {
|
||||||
|
const leaid = row.leaid
|
||||||
|
if (!leaid) continue
|
||||||
|
const prev = byLeaid.get(leaid) || { leaid, name: row.lea_name || leaid, arrests: 0, enrollment: 0 }
|
||||||
|
byLeaid.set(leaid, {
|
||||||
|
...prev,
|
||||||
|
name: prev.name || row.lea_name || leaid,
|
||||||
|
arrests: prev.arrests + (row.observed_arrests || 0),
|
||||||
|
enrollment: prev.enrollment + (row.stu_enroll || 0),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
const ranked = [...byLeaid.values()]
|
||||||
|
.filter((d) => d.arrests > 0)
|
||||||
|
.sort((a, b) => b.arrests - a.arrests || a.leaid.localeCompare(b.leaid))
|
||||||
|
.slice(0, TOP_N)
|
||||||
|
.map((d) => ({
|
||||||
|
leaid: d.leaid,
|
||||||
|
name: d.name,
|
||||||
|
arrests: d.arrests,
|
||||||
|
enrollment: d.enrollment,
|
||||||
|
rate: d.enrollment > 0 ? Math.round((d.arrests / d.enrollment) * 1000 * 100) / 100 : 0,
|
||||||
|
}))
|
||||||
|
|
||||||
|
return { state, districts: ranked, districtsSeen: byLeaid.size, rows: total, pages }
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Small fixed-size worker pool — polite to a single public API host. */
|
||||||
|
async function mapWithConcurrency(items, limit, worker) {
|
||||||
|
const results = new Array(items.length)
|
||||||
|
let next = 0
|
||||||
|
const runners = Array.from({ length: Math.min(limit, items.length) }, async () => {
|
||||||
|
while (next < items.length) {
|
||||||
|
const i = next++
|
||||||
|
results[i] = await worker(items[i], i)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
await Promise.all(runners)
|
||||||
|
return results
|
||||||
|
}
|
||||||
|
|
||||||
|
async function main() {
|
||||||
|
const states = parseStates()
|
||||||
|
console.log(`Fetching ${states.length} state(s) from ${BASE_URL} (year ${YEAR}, model ${MODEL})…`)
|
||||||
|
|
||||||
|
let done = 0
|
||||||
|
let requests = 0
|
||||||
|
const collected = await mapWithConcurrency(states, CONCURRENCY, async (state) => {
|
||||||
|
const result = await collectState(state)
|
||||||
|
requests += result.pages
|
||||||
|
done += 1
|
||||||
|
console.log(
|
||||||
|
` [${String(done).padStart(2)}/${states.length}] ${state}: ` +
|
||||||
|
`${result.rows} rows / ${result.pages} page(s), ` +
|
||||||
|
`${result.districtsSeen} districts, top ${result.districts.length} kept`,
|
||||||
|
)
|
||||||
|
return result
|
||||||
|
})
|
||||||
|
|
||||||
|
const byState = {}
|
||||||
|
for (const { state, districts } of collected.sort((a, b) => a.state.localeCompare(b.state))) {
|
||||||
|
byState[state] = districts
|
||||||
|
}
|
||||||
|
|
||||||
|
const payload = {
|
||||||
|
metadata: {
|
||||||
|
source: 'CRDC School Arrest Rate API (Knowles & Miller 2025)',
|
||||||
|
endpoint: `${BASE_URL}/estimates`,
|
||||||
|
year: YEAR,
|
||||||
|
model: MODEL,
|
||||||
|
description:
|
||||||
|
`Top ${TOP_N} school districts per state by total observed arrests in ${YEAR}, ` +
|
||||||
|
'summed across the eight modelled race×sex groups. Generated by ' +
|
||||||
|
'scripts/build-top-districts.mjs over the complete paged result set for each state.',
|
||||||
|
generated_states: states.length,
|
||||||
|
generated_requests: requests,
|
||||||
|
},
|
||||||
|
states: byState,
|
||||||
|
}
|
||||||
|
|
||||||
|
await mkdir(dirname(OUT_PATH), { recursive: true })
|
||||||
|
await writeFile(OUT_PATH, `${JSON.stringify(payload, null, 2)}\n`, 'utf8')
|
||||||
|
console.log(`\nWrote ${OUT_PATH} (${requests} requests, ${states.length} states).`)
|
||||||
|
}
|
||||||
|
|
||||||
|
main().catch((err) => {
|
||||||
|
console.error('\nbuild-top-districts failed:', err.message)
|
||||||
|
process.exit(1)
|
||||||
|
})
|
||||||
@@ -1,10 +1,42 @@
|
|||||||
import { useState, useEffect } from 'react'
|
import { useState, useEffect } from 'react'
|
||||||
import { searchDistricts, fetchStateDistricts } from '../hooks/useApi.js'
|
import { searchDistricts } from '../hooks/useApi.js'
|
||||||
|
|
||||||
|
const TOP_DISTRICTS_URL = `${import.meta.env.BASE_URL}data/top_districts.json`
|
||||||
|
const SUGGESTION_COUNT = 8
|
||||||
|
|
||||||
|
// Module-level cache: the fixture covers every state, so it is fetched at most
|
||||||
|
// once per page load no matter how many states the visitor browses through.
|
||||||
|
let topDistrictsPromise = null
|
||||||
|
|
||||||
|
function loadTopDistricts() {
|
||||||
|
if (!topDistrictsPromise) {
|
||||||
|
topDistrictsPromise = fetch(TOP_DISTRICTS_URL)
|
||||||
|
.then((res) => {
|
||||||
|
if (!res.ok) throw new Error(`HTTP ${res.status} loading ${TOP_DISTRICTS_URL}`)
|
||||||
|
return res.json()
|
||||||
|
})
|
||||||
|
.catch((err) => {
|
||||||
|
// Don't poison the cache with a rejected promise — let a later visit retry.
|
||||||
|
topDistrictsPromise = null
|
||||||
|
throw err
|
||||||
|
})
|
||||||
|
}
|
||||||
|
return topDistrictsPromise
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* District search screen with:
|
* District search screen with:
|
||||||
* 1. "Interesting" suggestions — districts with the most arrests in the selected state (fetched once)
|
* 1. "Interesting" suggestions — the districts with the most arrests in the
|
||||||
|
* selected state, read from the committed `public/data/top_districts.json`
|
||||||
|
* fixture (built by `scripts/build-top-districts.mjs`).
|
||||||
* 2. Live-search as you type → /api/v1/districts?q=...&state=XX
|
* 2. Live-search as you type → /api/v1/districts?q=...&state=XX
|
||||||
|
*
|
||||||
|
* The suggestions used to be computed at runtime from
|
||||||
|
* `/estimates?state=XX&year=21-22&limit=500`. That endpoint returns rows
|
||||||
|
* `ORDER BY LEAID, RACE, SEX` at eight rows per district, so the cap selected
|
||||||
|
* the ~62 lowest-LEAID districts in the state rather than the busiest ones —
|
||||||
|
* California has 11,488 rows. The fixture is ranked over the complete result
|
||||||
|
* set, and it also removes a multi-second fetch from this screen.
|
||||||
*/
|
*/
|
||||||
export default function DistrictSearch({ state, onSelect, onBack }) {
|
export default function DistrictSearch({ state, onSelect, onBack }) {
|
||||||
const [query, setQuery] = useState('')
|
const [query, setQuery] = useState('')
|
||||||
@@ -13,38 +45,32 @@ export default function DistrictSearch({ state, onSelect, onBack }) {
|
|||||||
const [loadingSugg, setLoadingSugg] = useState(true)
|
const [loadingSugg, setLoadingSugg] = useState(true)
|
||||||
const [loadingSearch, setLoadingSearch] = useState(false)
|
const [loadingSearch, setLoadingSearch] = useState(false)
|
||||||
|
|
||||||
// ——— Fetch interesting suggestions (top arrests in this state) once on mount ———
|
// ——— Read the suggestion fixture for this state ———
|
||||||
useEffect(() => {
|
useEffect(() => {
|
||||||
let cancelled = false
|
let cancelled = false
|
||||||
async function loadSuggestions() {
|
setLoadingSugg(true)
|
||||||
try {
|
|
||||||
// /estimates?state=XX&year=21-22 returns all districts; sort by observed_arrests desc client-side
|
|
||||||
const data = await fetchStateDistricts(state, '21-22', 500)
|
|
||||||
|
|
||||||
// Aggregate arrests per district (sum across race×sex groups), then sort
|
loadTopDistricts()
|
||||||
const byLeaid = {}
|
.then((fixture) => {
|
||||||
for (const row of data) {
|
if (cancelled) return
|
||||||
if (!byLeaid[row.leaid]) {
|
const forState = fixture?.states?.[state] || []
|
||||||
byLeaid[row.leaid] = { leaid: row.leaid, lea_name: row.lea_name, state: row.state, observed_arrests: 0 }
|
setSuggestions(
|
||||||
}
|
forState.slice(0, SUGGESTION_COUNT).map((d) => ({
|
||||||
byLeaid[row.leaid].observed_arrests += (row.observed_arrests || 0)
|
leaid: d.leaid,
|
||||||
}
|
lea_name: d.name,
|
||||||
|
state,
|
||||||
const sorted = Object.values(byLeaid).sort((a, b) => b.observed_arrests - a.observed_arrests)
|
observed_arrests: d.arrests,
|
||||||
|
})),
|
||||||
if (!cancelled) {
|
)
|
||||||
// Take top 8 for suggestions; filter to those with >0 arrests
|
})
|
||||||
setSuggestions(sorted.filter(d => d.observed_arrests > 0).slice(0, 8))
|
.catch((err) => {
|
||||||
}
|
console.error('Failed to load suggested districts:', err)
|
||||||
} catch (err) {
|
|
||||||
console.error('Failed to load interesting districts:', err)
|
|
||||||
if (!cancelled) setSuggestions([]) // degrade gracefully — just show search box
|
if (!cancelled) setSuggestions([]) // degrade gracefully — just show search box
|
||||||
} finally {
|
})
|
||||||
|
.finally(() => {
|
||||||
if (!cancelled) setLoadingSugg(false)
|
if (!cancelled) setLoadingSugg(false)
|
||||||
}
|
})
|
||||||
}
|
|
||||||
|
|
||||||
loadSuggestions()
|
|
||||||
return () => { cancelled = true }
|
return () => { cancelled = true }
|
||||||
}, [state])
|
}, [state])
|
||||||
|
|
||||||
@@ -158,11 +184,6 @@ export default function DistrictSearch({ state, onSelect, onBack }) {
|
|||||||
</div>
|
</div>
|
||||||
)}
|
)}
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
{/* Hint */}
|
|
||||||
<p style={{ fontSize: '0.85rem', color: 'var(--cv-ink-3)', marginTop: 'var(--space-4)' }}>
|
|
||||||
Tip: Try districts like Derby (KS), Paterson (NJ), or Mobile County (AL) — they have notable arrest rates.
|
|
||||||
</p>
|
|
||||||
</div>
|
</div>
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|||||||
+10
-1
@@ -119,7 +119,16 @@ export async function fetchStateEstimates(state, options = {}) {
|
|||||||
return apiFetch(`/states/${state}${qs ? `?${qs}` : ''}`)
|
return apiFetch(`/states/${state}${qs ? `?${qs}` : ''}`)
|
||||||
}
|
}
|
||||||
|
|
||||||
/** GET /estimates?state=XX&year=Y — all districts in a state for "interesting" suggestions */
|
/**
|
||||||
|
* GET /estimates?state=XX&year=Y — every estimate row in a state.
|
||||||
|
*
|
||||||
|
* Not used by the running app. The rows come back `ORDER BY LEAID, RACE, SEX`
|
||||||
|
* at eight per district, so any `limit` short of the state's full row count
|
||||||
|
* selects the lowest-LEAID districts rather than a meaningful sample —
|
||||||
|
* `DistrictSearch` reads the pre-ranked `public/data/top_districts.json`
|
||||||
|
* fixture instead. Kept for scripts and ad-hoc use; page it with `meta.total`
|
||||||
|
* as `scripts/build-top-districts.mjs` does.
|
||||||
|
*/
|
||||||
export async function fetchStateDistricts(state, year = '21-22', limit = 500) {
|
export async function fetchStateDistricts(state, year = '21-22', limit = 500) {
|
||||||
return apiFetch(`/estimates?state=${state}&year=${year}&limit=${limit}`)
|
return apiFetch(`/estimates?state=${state}&year=${year}&limit=${limit}`)
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user