Deploy to git-pages / deploy (push) Successful in 19s
The previous cap of 30 was exceeded by over 60% of districts — mostly sparse groups with small enrollment cells whose Agresti-Coull upper bounds genuinely extend that far. Analysis across all 51 states (16,279 districts) showed the median max x-value is already ~56 per 1,000; only truly degenerate cases like a single predicted arrest in a four-student cell (~250/1000) need capping. A cap of 100 still guards against these outliers while letting realistic data drive the axis for the vast majority of districts. The existing 'clipped' flag and note mechanism remain unchanged — they activate only when extreme values are encountered.
393 lines
16 KiB
JavaScript
393 lines
16 KiB
JavaScript
/**
|
||
* Analysis: distribution of maximum x-axis values for the rate density chart.
|
||
*
|
||
* The "Arrest rate probability density" chart currently caps its x-axis at a
|
||
* hard-coded MAX_RATE_DOMAIN = 30 (per 1,000 students). We want to understand
|
||
* what the true distribution of max x-values is across districts so we can
|
||
* decide whether to raise or make this cap dynamic.
|
||
*
|
||
* What feeds computeRateDomain():
|
||
* 1. Per-group rate arrays from real posterior draws → 0.995 quantile of each
|
||
* (requires parquet data; approximated here via model estimates)
|
||
* 2. Agresti-Coull upper bound rates = (ac.upper / enroll) * 1000, computed
|
||
* from observed counts and enrollment — available via the API for every
|
||
* race×sex cell in every district
|
||
*
|
||
* KEY INSIGHT: Only SELECTED groups feed into computeRateDomain. Per
|
||
* `defaultSelectedKeys` in districtGroups.js, only groups with ≥1 observed
|
||
* arrest are selected by default (or top-2 by enrollment if none have arrests).
|
||
* Additionally, when total district arrests < 20, sex pooling merges F+M cells,
|
||
* which combines enrollment and observed counts per race.
|
||
*/
|
||
|
||
const BASE = 'https://crdc-api.civilytics.org/api/v1'
|
||
const LIMIT = 500
|
||
const POOL_THRESHOLD = 20 // POOL_BY_SEX_ARREST_THRESHOLD from pooling.js
|
||
|
||
/** Fetch JSON from the API, retrying on transient failures. */
|
||
async function apiFetch(path) {
|
||
const url = `${BASE}${path}`
|
||
let lastErr
|
||
for (let attempt = 0; attempt <= 3; attempt++) {
|
||
try {
|
||
const res = await fetch(url, { signal: AbortSignal.timeout(30000) })
|
||
if (!res.ok) throw new Error(`HTTP ${res.status}`)
|
||
return await res.json()
|
||
} catch (err) {
|
||
lastErr = err
|
||
if (attempt < 3) {
|
||
const wait = 500 * Math.pow(2, attempt) + Math.random() * 100
|
||
console.error(` retry ${attempt + 1}/3 after ${Math.round(wait)}ms:`, err.message)
|
||
await new Promise((r) => setTimeout(r, wait))
|
||
}
|
||
}
|
||
}
|
||
throw lastErr
|
||
}
|
||
|
||
/** Peter Acklam's inverse normal CDF (probit), ~1.15e-9 relative error. */
|
||
function probit(p) {
|
||
const a = [-3.969683028665376e+01, 2.209460984245205e+02, -2.759285104469687e+02,
|
||
1.383577518672690e+02, -3.066479806614716e+01, 2.506628277459239e+00]
|
||
const b = [-5.447609879822406e+01, 1.615858368580409e+02, -1.556989798598866e+02,
|
||
6.680131188771972e+01, -1.328068155288572e+01]
|
||
const c = [-7.784894002430293e-03, -3.223964580411365e-01, -2.400758277161838e+00,
|
||
-2.549732539343734e+00, 4.374664141464968e+00, 2.938163982698783e+00]
|
||
const d = [7.784695709041462e-03, 3.224671290700398e-01, 2.445134137142996e+00,
|
||
3.754408661907416e+00]
|
||
const pLow = 0.02425, pHigh = 1 - pLow
|
||
|
||
if (p < pLow) {
|
||
const q = Math.sqrt(-2 * Math.log(p))
|
||
return (((((c[0]*q+c[1])*q+c[2])*q+c[3])*q+c[4])*q+c[5]) /
|
||
((((d[0]*q+d[1])*q+d[2])*q+d[3])*q+1)
|
||
}
|
||
if (p <= pHigh) {
|
||
const q = p - 0.5, r = q * q
|
||
return (((((a[0]*r+a[1])*r+a[2])*r+a[3])*r+a[4])*r+a[5]) * q /
|
||
(((((b[0]*r+b[1])*r+b[2])*r+b[3])*r+b[4])*r+1)
|
||
}
|
||
const q = Math.sqrt(-2 * Math.log(1 - p))
|
||
return -(((((c[0]*q+c[1])*q+c[2])*q+c[3])*q+c[4])*q+c[5]) /
|
||
((((d[0]*q+d[1])*q+d[2])*q+d[3])*q+1)
|
||
}
|
||
|
||
/** Agresti-Coull upper bound (count scale), port of agrestiCoull.js. */
|
||
function acUpperBound(numerator, denominator, confidenceLevel = 0.95) {
|
||
const adjStar = probit(1 - (1 - confidenceLevel) / 2)
|
||
if (numerator > 0) {
|
||
const numStar = numerator + adjStar
|
||
const denomStar = denominator + 2 * adjStar
|
||
const phat = numStar / denomStar
|
||
const se = Math.sqrt((phat / denomStar) * (1 - phat))
|
||
return (phat + adjStar * se) * denomStar
|
||
}
|
||
// Zero events: rule of three — upper bound ≈ 3 regardless of enrollment.
|
||
return -Math.log(1 - confidenceLevel)
|
||
}
|
||
|
||
/** Estimate the 0.995 quantile of posterior predictive draw rates. */
|
||
function estimateDrawQuantile(row, targetP = 0.995) {
|
||
const enroll = row.stu_enroll || 0
|
||
if (enroll <= 0) return null
|
||
|
||
// When count_upper is available and > 0, the model's posterior predictive
|
||
// upper bound gives a sense of the spread. For sparse groups with few or zero
|
||
// arrests, draw quantiles can be extreme because observation noise dominates:
|
||
// a single predicted arrest in a small cell produces a huge per-1000 rate.
|
||
const countUpper = row.count_upper || 0
|
||
if (countUpper > 0) {
|
||
return (countUpper / enroll) * 1000
|
||
}
|
||
|
||
// When count_upper is 0, use the Agresti-Coull upper bound as a conservative
|
||
// proxy — it's an honest frequentist interval that also tends to be extreme
|
||
// for sparse groups.
|
||
const ac = acUpperBound(row.observed_arrests || 0, enroll)
|
||
return (ac / enroll) * 1000
|
||
}
|
||
|
||
/** Determine which race×sex cells are "selected" by default per districtGroups.js logic. */
|
||
function determineSelectedCells(cells, pooled) {
|
||
const usable = cells.filter(
|
||
(r) => ['WH', 'BL', 'HI', 'AM'].includes(r.race) && ['F', 'M'].includes(r.sex),
|
||
)
|
||
|
||
if (pooled) {
|
||
// When pooled: merge F+M per race, then select groups with ≥1 arrest or top-2 by enrollment
|
||
const races = {}
|
||
for (const r of usable) {
|
||
if (!races[r.race]) races[r.race] = { race: r.race, enroll: 0, observed: 0 }
|
||
races[r.race].enroll += r.stu_enroll || 0
|
||
races[r.race].observed += r.observed_arrests || 0
|
||
}
|
||
const merged = Object.values(races)
|
||
|
||
// defaultSelectedKeys logic on pooled groups
|
||
const withArrests = merged.filter((r) => r.observed > 0)
|
||
if (withArrests.length > 0) {
|
||
return withArrests.map((r) => ({ race: r.race, sex: null, enroll: r.enroll, observed: r.observed }))
|
||
}
|
||
return [...merged]
|
||
.sort((a, b) => b.enroll - a.enroll)
|
||
.slice(0, 2)
|
||
.map((r) => ({ race: r.race, sex: null, enroll: r.enroll, observed: r.observed }))
|
||
} else {
|
||
// Unpooled: select groups with ≥1 arrest or top-2 by enrollment
|
||
const withArrests = usable.filter((r) => (r.observed_arrests || 0) > 0)
|
||
if (withArrests.length > 0) return withArrests
|
||
|
||
return [...usable]
|
||
.sort((a, b) => (b.stu_enroll || 0) - (a.stu_enroll || 0))
|
||
.slice(0, 2)
|
||
}
|
||
}
|
||
|
||
/** Compute max x-value candidate from selected cells. */
|
||
function computeMaxX(selectedCells, allCells, pooled) {
|
||
let maxAcRate = 0
|
||
let maxDrawEstimate = 0
|
||
|
||
for (const cell of selectedCells) {
|
||
const enroll = cell.enroll || 0
|
||
if (enroll <= 0) continue
|
||
const observed = cell.observed || 0
|
||
|
||
// AC upper bound rate (always a candidate in computeRateDomain)
|
||
const acUpper = acUpperBound(observed, enroll)
|
||
const acRate = (acUpper / enroll) * 1000
|
||
if (acRate > maxAcRate) maxAcRate = acRate
|
||
|
||
// Estimated draw quantile — find the matching row in allCells for model estimates
|
||
let countUpper = 0
|
||
let foundRow = null
|
||
if (!pooled && cell.sex) {
|
||
foundRow = allCells.find((r) => r.race === cell.race && r.sex === cell.sex)
|
||
} else if (pooled) {
|
||
// For pooled, find the row with max count_upper for this race across both sexes
|
||
const raceRows = allCells.filter((r) => r.race === cell.race && ['F', 'M'].includes(r.sex))
|
||
for (const rr of raceRows) {
|
||
if ((rr.count_upper || 0) > countUpper) countUpper = rr.count_upper || 0
|
||
}
|
||
}
|
||
|
||
if (!pooled && foundRow) {
|
||
countUpper = foundRow.count_upper || 0
|
||
}
|
||
|
||
const enroll_ = cell.enroll || 1
|
||
let drawEst = null
|
||
if (countUpper > 0) {
|
||
drawEst = (countUpper / enroll_) * 1000
|
||
} else {
|
||
// Conservative proxy via AC upper bound
|
||
const ac = acUpperBound(observed, enroll_)
|
||
drawEst = (ac / enroll_) * 1000
|
||
}
|
||
|
||
if (drawEst > maxDrawEstimate) maxDrawEstimate = drawEst
|
||
}
|
||
|
||
return Math.max(maxAcRate, maxDrawEstimate)
|
||
}
|
||
|
||
async function main() {
|
||
console.log('=== Rate Domain Analysis ===\n')
|
||
|
||
// Step 1: Get all states from the API
|
||
const statesResp = await apiFetch('/states?limit=500')
|
||
const stateSet = new Set(statesResp.data.map((r) => r.state))
|
||
const states = [...stateSet].sort()
|
||
console.log(`Found ${states.length} states\n`)
|
||
|
||
// Step 2: For each state, page through all estimates and collect per-district data
|
||
const districtData = new Map() // leaid -> { state, name, cells: [] }
|
||
let totalRows = 0
|
||
|
||
for (const state of states) {
|
||
console.log(`Fetching ${state}...`)
|
||
const firstResp = await apiFetch(`/estimates?state=${state}&limit=${LIMIT}`)
|
||
const total = firstResp.meta.total
|
||
const nPages = Math.ceil(total / LIMIT)
|
||
|
||
let rows = [...firstResp.data]
|
||
for (let page = 1; page < nPages; page++) {
|
||
process.stderr.write(` ${state} page ${page + 1}/${nPages}\r`)
|
||
const resp = await apiFetch(`/estimates?state=${state}&limit=${LIMIT}&page=${page}`)
|
||
rows.push(...resp.data)
|
||
}
|
||
totalRows += rows.length
|
||
|
||
for (const row of rows) {
|
||
const key = `${row.state}|${row.leaid}`
|
||
if (!districtData.has(key)) {
|
||
districtData.set(key, { state: row.state, leaid: row.leaid, name: row.lea_name, cells: [] })
|
||
}
|
||
districtData.get(key).cells.push(row)
|
||
}
|
||
}
|
||
|
||
console.log(`\nTotal rows fetched: ${totalRows}`)
|
||
console.log(`Total districts: ${districtData.size}\n`)
|
||
|
||
// Step 3: For each district, compute the max x-value for both pooled and unpooled modes
|
||
const results = []
|
||
let nPooled = 0
|
||
let nUnpooled = 0
|
||
|
||
for (const [key, dist] of districtData) {
|
||
const totalArrests = dist.cells.reduce((sum, r) => sum + (r.observed_arrests || 0), 0)
|
||
const pooled = totalArrests < POOL_THRESHOLD
|
||
if (pooled) nPooled++
|
||
else nUnpooled++
|
||
|
||
let maxX
|
||
if (pooled) {
|
||
// Pooled mode: AC bounds computed per race (F+M merged). Note that for
|
||
// pooled groups, the app shows a note but still computes AC bounds.
|
||
const selected = determineSelectedCells(dist.cells, true)
|
||
maxX = computeMaxX(selected, dist.cells, true)
|
||
} else {
|
||
const selected = determineSelectedCells(dist.cells, false)
|
||
maxX = computeMaxX(selected, dist.cells, false)
|
||
}
|
||
|
||
let totalEnroll = 0
|
||
for (const cell of dist.cells) {
|
||
if ((cell.stu_enroll || 0) > 0) totalEnroll += cell.stu_enroll
|
||
}
|
||
|
||
results.push({
|
||
key, state: dist.state, leaid: dist.leaid, name: dist.name,
|
||
totalEnroll, pooled, maxX: maxX * 1.15, // HEADROOM factor from rateDomain.js
|
||
})
|
||
}
|
||
|
||
console.log(`Districts with sex pooling (total arrests < ${POOL_THRESHOLD}): ${nPooled} (${(nPooled / results.length * 100).toFixed(1)}%)`)
|
||
console.log(`Districts without pooling: ${nUnpooled} (${(nUnpooled / results.length * 100).toFixed(1)}%)\n`)
|
||
|
||
// Step 4: Analyze distribution of max x-values
|
||
const sorted = results.sort((a, b) => a.maxX - b.maxX)
|
||
const n = sorted.length
|
||
|
||
console.log('=== Distribution of Maximum X-Values (per 1,000 students) ===\n')
|
||
|
||
// Summary statistics — maxX already includes HEADROOM(1.15) factor
|
||
const percentiles = [5, 10, 25, 50, 75, 90, 95, 99, 99.9]
|
||
console.log('Percentiles of max x-value (includes HEADROOM=1.15):')
|
||
for (const p of percentiles) {
|
||
const idx = Math.floor((p / 100) * (n - 1))
|
||
console.log(` ${p.toFixed(1)}th: ${sorted[idx].maxX.toFixed(2)} per 1,000`)
|
||
}
|
||
|
||
console.log('\n--- Threshold analysis ---')
|
||
const thresholds = [30, 40, 50, 60, 70, 80, 90, 100]
|
||
for (const thresh of thresholds) {
|
||
const count = sorted.filter((r) => r.maxX > thresh).length
|
||
const pct = (count / n) * 100
|
||
console.log(` Exceeds ${thresh}: ${count} districts (${pct.toFixed(2)}%)`)
|
||
}
|
||
|
||
// Step 5: Break down by pooling status
|
||
console.log('\n--- By pooling status ---')
|
||
const pooledResults = sorted.filter((r) => r.pooled)
|
||
const unpooledResults = sorted.filter((r) => !r.pooled)
|
||
for (const thresh of [30, 50, 100]) {
|
||
const pClipped = pooledResults.filter((r) => r.maxX > thresh).length
|
||
const uClipped = unpooledResults.filter((r) => r.maxX > thresh).length
|
||
console.log(` Cap=${thresh}: pooled ${pClipped}/${pooledResults.length} (${(pClipped / pooledResults.length * 100).toFixed(1)}%), ` +
|
||
`unpooled ${uClipped}/${unpooledResults.length} (${(uClipped / unpooledResults.length * 100).toFixed(1)}%)`)
|
||
}
|
||
|
||
// Step 6: Show top districts by max x-value, with enrollment context
|
||
console.log('\n--- Top 25 districts by max x-value ---')
|
||
const top = sorted.slice(-25).reverse()
|
||
for (const r of top) {
|
||
const pooledStr = r.pooled ? ' [pooled]' : ''
|
||
const clippedAt30 = r.maxX > 30 ? ' *** CLIPPED at 30' : ''
|
||
console.log(` ${r.state} | LEAID ${r.leaid} | enroll=${r.totalEnroll.toLocaleString()}${pooledStr} | ` +
|
||
`max_x=${r.maxX.toFixed(2)}/1000` + clippedAt30)
|
||
}
|
||
|
||
// Step 7: Show realistic-size districts (enrollment > 500) that are clipped
|
||
console.log('\n--- Clipped districts with enrollment > 500 ---')
|
||
const realClipped = sorted.filter((r) => r.maxX > 30 && r.totalEnroll >= 500).sort((a, b) => b.maxX - a.maxX).slice(0, 15)
|
||
for (const r of realClipped) {
|
||
const pooledStr = r.pooled ? ' [pooled]' : ''
|
||
console.log(` ${r.state} | ${r.name} | enroll=${r.totalEnroll.toLocaleString()}${pooledStr} | ` +
|
||
`max_x=${r.maxX.toFixed(2)}/1000`)
|
||
}
|
||
|
||
// Step 8: Non-clipped districts for context
|
||
const notClipped = sorted.filter((r) => r.maxX <= 30).sort((a, b) => a.maxX - b.maxX)
|
||
console.log(`\n--- Non-clipped districts (max_x ≤ 30): ${notClipped.length} (${(notClipped.length / n * 100).toFixed(2)}%) ---`)
|
||
if (notClipped.length > 0) {
|
||
const midIdx = Math.floor(notClipped.length / 2)
|
||
console.log(' Sample non-clipped districts:')
|
||
for (let i = Math.max(0, midIdx - 3); i < Math.min(notClipped.length, midIdx + 4); i++) {
|
||
const r = notClipped[i]
|
||
console.log(` ${r.state} | ${r.name} | enroll=${r.totalEnroll.toLocaleString()} | max_x=${r.maxX.toFixed(2)}/1000`)
|
||
}
|
||
}
|
||
|
||
// Step 9: State-level summary (focusing on clipped counts)
|
||
console.log('\n--- Top 15 states by % of districts clipped ---')
|
||
const byState = {}
|
||
for (const r of results) {
|
||
if (!byState[r.state]) byState[r.state] = []
|
||
byState[r.state].push(r.maxX)
|
||
}
|
||
const stateStats = Object.entries(byState).map(([state, vals]) => ({
|
||
state, n: vals.length, median: percentile(vals, 50), p95: percentile(vals, 95),
|
||
max: Math.max(...vals), clipped: vals.filter((v) => v > 30).length,
|
||
})).sort((a, b) => (b.clipped / b.n) - (a.clipped / a.n))
|
||
|
||
for (const s of stateStats.slice(0, 15)) {
|
||
console.log(` ${s.state}: n=${s.n}, median=${s.median.toFixed(1)}, p95=${s.p95.toFixed(1)}, ` +
|
||
`max=${s.max.toFixed(1)}, clipped>30: ${s.clipped} (${(s.clipped / s.n * 100).toFixed(1)}%)`)
|
||
}
|
||
|
||
// Step 10: Recommendation analysis — what cap would minimize clipping while staying bounded?
|
||
console.log('\n=== RECOMMENDATION ANALYSIS ===')
|
||
const capOptions = [30, 40, 50, 60, 75, 100]
|
||
for (const cap of capOptions) {
|
||
const clipped = sorted.filter((r) => r.maxX > cap).length
|
||
console.log(` Cap=${cap}: ${clipped} districts clipped (${(clipped / n * 100).toFixed(2)}%)`)
|
||
}
|
||
|
||
// Step 11: Key findings summary
|
||
console.log('\n--- Key Findings ---')
|
||
const pctExceed30 = (sorted.filter((r) => r.maxX > 30).length / n) * 100
|
||
const pctExceed50 = (sorted.filter((r) => r.maxX > 50).length / n) * 100
|
||
console.log(`- ${pctExceed30.toFixed(2)}% of districts have a max x-value exceeding the current cap of 30`)
|
||
console.log(`- ${pctExceed50.toFixed(2)}% exceed 50 per 1,000`)
|
||
const p99 = sorted[Math.floor(0.99 * (n - 1))].maxX
|
||
const max = sorted[n - 1].maxX
|
||
console.log(`- 99th percentile: ${p99.toFixed(2)} per 1,000`)
|
||
console.log(`- Maximum observed: ${max.toFixed(2)} per 1,000`)
|
||
|
||
// Analyze the nature of clipped districts — are they sparse or not?
|
||
const extreme = sorted.filter((r) => r.maxX > 30).sort((a, b) => a.totalEnroll - b.totalEnroll)
|
||
console.log(`\n--- Clipped district enrollment distribution ---`)
|
||
for (const p of [10, 25, 50, 75, 90]) {
|
||
const idx = Math.floor((p / 100) * (extreme.length - 1))
|
||
console.log(` ${p}th percentile enrollment: ${extreme[idx].totalEnroll.toLocaleString()}`)
|
||
}
|
||
|
||
// How many clipped districts have "normal" school sizes (>1000 students)?
|
||
const normalClipped = extreme.filter((r) => r.totalEnroll >= 1000).length
|
||
console.log(`\n- ${normalClipped} of ${extreme.length} clipped districts have enrollment ≥ 1,000 (${(normalClipped / extreme.length * 100).toFixed(1)}%)`)
|
||
|
||
// Analyze what's driving the extremes — AC bounds vs draw estimates
|
||
console.log('\n--- What drives extreme values? ---')
|
||
const acDriven = sorted.filter((r) => r.maxX > 30 && !r.pooled).length
|
||
console.log(`- Unpooled districts clipped: ${acDriven} (${(acDriven / n * 100).toFixed(2)}% of all)`)
|
||
|
||
function percentile(arr, p) {
|
||
const s = [...arr].sort((a, b) => a - b)
|
||
return s[Math.floor((p / 100) * (s.length - 1))]
|
||
}
|
||
}
|
||
|
||
main().catch(console.error)
|