admin: STT eval harness — record human-verified transcripts, measure real WER (#163)
Build & Deploy / Build & push images (push) Successful in 4m10s
Build & Deploy / Deploy Firestore rules & indexes (push) Failing after 3s
Build & Deploy / Deploy to VM (push) Successful in 1m56s
Build & Deploy / Report a failed deploy (push) Successful in 1s

Backend: three new routes on the calls router, deliberately separate from
PATCH /{call_id}/transcript (a production correction with real side effects
-- re-extraction, incident unlinking, vocabulary learning). This is pure
measurement and must never share that code path.

  GET  /calls/eval-queue        -- calls with a transcript but no
                                    eval_transcript yet, paged (same bounded-
                                    window-plus-cursor shape as /search)
  PUT  /{call_id}/eval-transcript -- records eval_transcript/_by/_at only;
                                      never touches transcript/transcript_corrected
  GET  /calls/eval-stats        -- eval_count + average word error rate of
                                    the raw and corrected machine transcripts
                                    against the human-verified ones

internal/wer.py: standard word-level Levenshtein WER. Returns None (not 0.0)
when the reference is empty -- a call nobody transcribed must not score as a
perfect match.

Frontend: a new "STT Eval" tab on /admin -- one call at a time, audio player,
a textarea pre-filled with the machine transcript to correct into ground
truth, Save & next / Skip, running WER stats at the top. Built for working a
handful of calls at a time over however many sittings it takes, not a
one-shot form: the queue auto-refills from where the last save left off.

Verified: 438 pass, 0 fail (12 new backend tests). Frontend is UNVERIFIED --
this box has no Node.js/npm (confirmed absent), so neither typecheck nor the
dev server could be run. Matches existing code patterns and the CallRecord/
c2api types by manual review only.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
Logan Cusano
2026-09-21 00:04:30 -04:00
co-authored by Claude Sonnet 5
parent 241a15b8da
commit 5f85a878fa
6 changed files with 530 additions and 2 deletions
+180 -2
View File
@@ -4,7 +4,7 @@ import { useAuth } from "@/components/AuthProvider";
import { c2api } from "@/lib/c2api";
import { useEffect, useState, useRef, useCallback } from "react";
import { useRouter } from "next/navigation";
import type { UserRecord, AuditEntry, UserRole } from "@/lib/types";
import type { UserRecord, AuditEntry, UserRole, CallRecord } from "@/lib/types";
// ---------------------------------------------------------------------------
// Shared primitives
@@ -1047,16 +1047,193 @@ function StaleCallsTab() {
);
}
// ---------------------------------------------------------------------------
// STT eval (server-26#163) — the eval harness for real transcription
// accuracy. Separate from patchTranscript's "fix this call" flow: this never
// re-runs extraction or touches an incident, it only records what was
// actually said next to what Whisper heard, so eval-stats can report a real
// WER instead of a guess. Built to be worked in short sessions, a handful of
// calls at a time, over however many sittings it takes — not a one-shot form.
// ---------------------------------------------------------------------------
function fmtPct(x: number | null | undefined): string {
return x === null || x === undefined ? "—" : `${(x * 100).toFixed(1)}%`;
}
function EvalStatsBar({ stats }: { stats: { eval_count: number; raw_wer: number | null; corrected_wer: number | null } | null }) {
return (
<div className="bg-gray-900 border border-gray-800 rounded-xl p-4 flex flex-wrap gap-x-8 gap-y-2">
<div>
<p className="text-xs text-gray-500 font-mono">Calls verified</p>
<p className="text-white text-lg font-mono">{stats?.eval_count ?? "—"}</p>
</div>
<div>
<p className="text-xs text-gray-500 font-mono">Raw WER (whisper-1)</p>
<p className="text-white text-lg font-mono">{fmtPct(stats?.raw_wer)}</p>
</div>
<div>
<p className="text-xs text-gray-500 font-mono">Corrected WER (shipped)</p>
<p className="text-white text-lg font-mono">{fmtPct(stats?.corrected_wer)}</p>
</div>
{stats && stats.eval_count > 0 && stats.eval_count < 20 && (
<p className="text-xs text-amber-400 font-mono self-end">
fewer than 20 calls — numbers will move a lot until this grows
</p>
)}
</div>
);
}
function SttEvalTab() {
const [stats, setStats] = useState<{ eval_count: number; raw_wer: number | null; corrected_wer: number | null } | null>(null);
const [queue, setQueue] = useState<CallRecord[]>([]);
const [cursor, setCursor] = useState<string | null>(null);
const [exhausted, setExhausted] = useState(false);
const [draft, setDraft] = useState("");
const [loadingBatch, setLoadingBatch] = useState(false);
const [saving, setSaving] = useState(false);
const [error, setError] = useState<string | null>(null);
const fetching = useRef(false);
const current = queue[0] ?? null;
const refreshStats = useCallback(() => {
c2api.getEvalStats().then(setStats).catch(() => { /* stats are a nice-to-have, not load-bearing */ });
}, []);
const loadBatch = useCallback(async () => {
if (fetching.current) return;
fetching.current = true;
setLoadingBatch(true);
setError(null);
try {
const res = await c2api.getEvalQueue(5, cursor);
setQueue((q) => [...q, ...res.calls]);
setCursor(res.next_cursor);
if (res.calls.length === 0 && !res.next_cursor) setExhausted(true);
} catch (e) {
setError(String(e));
} finally {
setLoadingBatch(false);
fetching.current = false;
}
}, [cursor]);
useEffect(() => { refreshStats(); }, [refreshStats]);
// Auto-refill: whenever the local queue runs dry and there's more to scan
// (or we haven't checked yet), pull another batch. Covers the sparse-window
// case too — a page with matches:0 but a next_cursor just means "keep
// scanning", not "done", so this fires again on its own.
useEffect(() => {
if (queue.length === 0 && !exhausted) loadBatch();
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [queue.length, exhausted]);
useEffect(() => {
setDraft(current ? (current.transcript_corrected || current.transcript || "") : "");
}, [current]);
async function saveAndNext() {
if (!current) return;
setSaving(true);
setError(null);
try {
await c2api.putEvalTranscript(current.call_id, draft);
setQueue((q) => q.slice(1));
refreshStats();
} catch (e) {
setError(String(e));
} finally {
setSaving(false);
}
}
function skip() {
setQueue((q) => q.slice(1));
}
return (
<div className="space-y-4">
<p className="text-xs text-gray-500 font-mono">
Listen to the audio, correct the transcript below until it matches what was actually said, then save.
This never touches the call&apos;s real transcript or re-runs anything — it only records ground truth
for measuring the pipeline. Do as many or as few as you have time for; it picks up where you left off.
</p>
<EvalStatsBar stats={stats} />
{error && (
<div className="bg-red-950 border border-red-800 rounded-lg p-3">
<p className="text-red-400 text-sm font-mono">{error}</p>
</div>
)}
{current ? (
<div className="bg-gray-900 border border-gray-800 rounded-xl p-4 space-y-3">
<div className="flex flex-wrap items-center gap-x-3 gap-y-1 text-xs font-mono text-gray-400">
<span>{new Date(current.started_at).toLocaleString()}</span>
<span>{current.talkgroup_name || (current.talkgroup_id ? `TGID ${current.talkgroup_id}` : "unknown talkgroup")}</span>
</div>
{current.audio_url ? (
// eslint-disable-next-line jsx-a11y/media-has-caption
<audio controls src={current.audio_url} className="w-full h-9" />
) : (
<p className="text-xs text-gray-500 italic">No audio on this call — skip it.</p>
)}
<div>
<label className="text-xs text-gray-400 block mb-1">
Machine transcript (pre-filled) — correct it into what was actually said
</label>
<textarea
value={draft}
onChange={(e) => setDraft(e.target.value)}
rows={4}
className="w-full bg-gray-800 border border-gray-700 rounded-lg px-3 py-2 text-white text-sm font-mono focus:outline-none focus:border-indigo-500"
/>
</div>
<div className="flex gap-2">
<button
onClick={saveAndNext}
disabled={saving || !draft.trim()}
className="bg-indigo-600 hover:bg-indigo-500 disabled:opacity-50 text-white text-sm font-mono px-4 py-1.5 rounded-lg transition-colors"
>
{saving ? "Saving…" : "Save & next"}
</button>
<button
onClick={skip}
disabled={saving}
className="bg-gray-800 hover:bg-gray-700 disabled:opacity-50 border border-gray-700 text-white text-sm font-mono px-4 py-1.5 rounded-lg transition-colors"
>
Skip
</button>
</div>
</div>
) : (
<div className="bg-gray-900 border border-gray-800 rounded-xl p-4">
<p className="text-sm font-mono text-gray-400">
{loadingBatch ? "Loading calls…" : exhausted ? "Nothing left to verify right now — check back after more calls come in." : "Loading…"}
</p>
</div>
)}
</div>
);
}
// ---------------------------------------------------------------------------
// Main admin page
// ---------------------------------------------------------------------------
type AdminTab = "features" | "correlation" | "users" | "audit" | "calls";
type AdminTab = "features" | "correlation" | "users" | "audit" | "calls" | "eval";
const TAB_LABELS: { key: AdminTab; label: string }[] = [
{ key: "features", label: "AI Features" },
{ key: "correlation", label: "Correlation Debug" },
{ key: "calls", label: "Calls" },
{ key: "eval", label: "STT Eval" },
{ key: "users", label: "Users" },
{ key: "audit", label: "Audit Log" },
];
@@ -1102,6 +1279,7 @@ export default function AdminPage() {
{tab === "features" && <FeaturesTab />}
{tab === "correlation" && <CorrelationDebugTab />}
{tab === "calls" && <StaleCallsTab />}
{tab === "eval" && <SttEvalTab />}
{tab === "users" && <UsersTab currentUid={user?.uid ?? ""} />}
{tab === "audit" && <AuditLogTab />}
</div>
+24
View File
@@ -100,6 +100,30 @@ export const c2api = {
closeStallCalls: (olderThanMinutes: number, dryRun: boolean) =>
request<{ dry_run: boolean; older_than_minutes: number; count: number; call_ids: string[] }>(`/calls/close-stale?older_than_minutes=${olderThanMinutes}&dry_run=${dryRun}`, { method: "POST" }),
// STT eval harness (server-26#163) — separate from patchTranscript above,
// which is a production correction with real side effects (re-extraction,
// incident unlinking, vocabulary learning). This is pure measurement.
getEvalQueue: (limit: number, cursor?: string | null) => {
const qs = new URLSearchParams({ limit: String(limit) });
if (cursor) qs.set("cursor", cursor);
return request<{
calls: import("@/lib/types").CallRecord[];
next_cursor: string | null;
scanned: number;
matched: number;
window_exhausted: boolean;
}>(`/calls/eval-queue?${qs.toString()}`);
},
getEvalStats: () =>
request<{ eval_count: number; raw_wer: number | null; corrected_wer: number | null }>(
"/calls/eval-stats"
),
putEvalTranscript: (callId: string, text: string) =>
request<{ ok: boolean; call_id: string }>(`/calls/${callId}/eval-transcript`, {
method: "PUT",
body: JSON.stringify({ text }),
}),
// Incidents
getIncidents: (params?: { status?: string; type?: string }) => {
const qs = params ? "?" + new URLSearchParams(params as Record<string, string>).toString() : "";
+4
View File
@@ -158,6 +158,10 @@ export interface CallRecord {
corr_incident_idle_min?: number | null;
corr_shared_units?: number | null;
corr_candidates?: number | null;
/** Human-verified reference transcript for the STT eval harness (server-26#163) — never read by anything downstream. */
eval_transcript?: string | null;
eval_transcript_by?: string | null;
eval_transcript_at?: string | null;
}
export interface IncidentRecord {