diff --git a/drb-c2-core/app/internal/wer.py b/drb-c2-core/app/internal/wer.py new file mode 100644 index 0000000..8d4d374 --- /dev/null +++ b/drb-c2-core/app/internal/wer.py @@ -0,0 +1,44 @@ +""" +Word error rate — server-26#163's eval harness needs a real number to compare +against, not a vibe. Standard definition: word-level Levenshtein distance +between a human-verified reference and the machine hypothesis, divided by the +reference's own word count. Case-insensitive, punctuation-insensitive — this +measures whether the right WORDS came out, not transcript formatting. +""" +import re +from typing import Optional + + +def _tokenize(text: str) -> list[str]: + return re.findall(r"[\w']+", (text or "").lower()) + + +def word_error_rate(reference: str, hypothesis: str) -> Optional[float]: + """ + (substitutions + deletions + insertions) / len(reference words). + + None when the reference has no words — WER is undefined there, not 0.0; + a caller that defaults a None to 0.0 would report a perfect score for a + call nobody actually transcribed. + """ + ref = _tokenize(reference) + hyp = _tokenize(hypothesis) + if not ref: + return None + if not hyp: + return 1.0 + + n, m = len(ref), len(hyp) + # Single-row DP over Levenshtein distance — O(n*m) time, O(m) space. + row = list(range(m + 1)) + for i in range(1, n + 1): + prev_diag = row[0] + row[0] = i + for j in range(1, m + 1): + prev_row_j = row[j] + if ref[i - 1] == hyp[j - 1]: + row[j] = prev_diag + else: + row[j] = 1 + min(prev_diag, row[j], row[j - 1]) + prev_diag = prev_row_j + return row[m] / n diff --git a/drb-c2-core/app/routers/calls.py b/drb-c2-core/app/routers/calls.py index b3efaa3..03caab4 100644 --- a/drb-c2-core/app/routers/calls.py +++ b/drb-c2-core/app/routers/calls.py @@ -15,6 +15,10 @@ from app.internal.storage import gcs_uri_for_call, with_playback_url class TranscriptUpdate(BaseModel): transcript: str + +class EvalTranscriptUpdate(BaseModel): + text: str + router = APIRouter(prefix="/calls", tags=["calls"]) @@ -130,6 +134,109 @@ async def search_calls( } +@router.get("/eval-queue") +async def eval_queue( + limit: int = Query(5, ge=1, le=20), + cursor: Optional[str] = Query(None, description="started_at of the last row of the previous page"), + decoded: dict = Depends(require_admin_token), +): + """ + A batch of calls that have a machine transcript but no human-verified one + yet — the backend for the STT eval page (server-26#163). + + Deliberately separate from `PATCH /{call_id}/transcript`: that route is a + PRODUCTION correction — it re-runs extraction, unlinks incidents, and + feeds the vocabulary learner. An eval annotation must never trigger any + of that; it only exists to measure the pipeline, not to change what it + already decided. `eval_transcript` lives next to `transcript`/ + `transcript_corrected` on the call doc and nothing downstream reads it. + + Same bounded-window-scan-plus-cursor shape as `/search`, for the same + reason: no composite index exists for "eval_transcript is unset", and one + scan ordered by started_at is already trusted here. Paging through with + the returned cursor is how "however many, over time" actually works — + each call is where the last session left off, not a fresh random sample. + """ + org_id = await resolve_caller_org_id(decoded) + if org_id is None: + org_id = decoded.get("org_id") + if not org_id: + raise HTTPException(403, "No organization scope for this caller.") + + window = max(limit * 20, 300) + rows = await fstore.collection_where( + "calls", + [("org_id", "==", org_id)], + order_by=[("started_at", "DESCENDING")], + limit_to=window, + start_after={"started_at": cursor} if cursor else None, + ) + + def _eligible(c: dict) -> bool: + text = c.get("transcript_corrected") or c.get("transcript") or "" + return bool(text) and not c.get("eval_transcript") + + matches = [c for c in rows if _eligible(c)] + page = matches[:limit] + + next_cursor = None + if len(rows) == window: + last_scanned = rows[-1].get("started_at") + next_cursor = last_scanned.isoformat() if hasattr(last_scanned, "isoformat") else last_scanned + + return { + "calls": [with_playback_url(c) for c in page], + "next_cursor": next_cursor, + "scanned": len(rows), + "matched": len(matches), + "window_exhausted": len(rows) == window, + } + + +@router.get("/eval-stats") +async def eval_stats(decoded: dict = Depends(require_admin_token)): + """ + How many calls have a human-verified transcript, and the WER of the raw + and corrected machine transcripts against them (server-26#163). + + Whole-collection scan, matching `GET /calls` (list_calls above) rather + than the bounded-window pattern the paged routes use: the eval set this + is measuring is built a few calls at a time and expected to stay small + (tens to hundreds), so a full scan filtered in Python is the honest + answer rather than a windowed guess that could miss eval'd calls sitting + outside a recency window. + """ + from app.internal.wer import word_error_rate + + org_id = await resolve_caller_org_id(decoded) + filters = {"org_id": org_id} if org_id is not None else {} + calls = await fstore.collection_list("calls", **filters) + + raw_wers: list[float] = [] + corrected_wers: list[float] = [] + for c in calls: + ref = c.get("eval_transcript") + if not ref: + continue + raw = c.get("transcript") or "" + corrected = c.get("transcript_corrected") or raw + raw_wer = word_error_rate(ref, raw) + corrected_wer = word_error_rate(ref, corrected) + if raw_wer is not None: + raw_wers.append(raw_wer) + if corrected_wer is not None: + corrected_wers.append(corrected_wer) + + def _avg(xs: list[float]) -> Optional[float]: + return round(sum(xs) / len(xs), 4) if xs else None + + return { + "eval_count": len(raw_wers), + "raw_wer": _avg(raw_wers), + "corrected_wer": _avg(corrected_wers), + } + + @router.get("/{call_id}") async def get_call(call_id: str, decoded: dict = Depends(require_service_or_firebase_token)): call = await fstore.doc_get("calls", call_id) @@ -313,3 +420,29 @@ async def patch_transcript( preserve_transcript_correction=True, ) return {"ok": True, "call_id": call_id} + + +@router.put("/{call_id}/eval-transcript") +async def put_eval_transcript( + call_id: str, + body: EvalTranscriptUpdate, + decoded: dict = Depends(require_admin_token), +): + """ + Record a human-verified reference transcript for the STT eval harness + (server-26#163). Pure data capture — unlike `PATCH /{call_id}/transcript` + above, this never touches `transcript`/`transcript_corrected`, never + re-runs extraction, never unlinks incidents, and never feeds the + vocabulary learner. It exists to MEASURE the pipeline's output, not to + change it; the two must not share a code path. + """ + call = await fstore.doc_get("calls", call_id) + if not call: + raise HTTPException(404, f"Call '{call_id}' not found.") + + await fstore.doc_set("calls", call_id, { + "eval_transcript": body.text, + "eval_transcript_by": decoded.get("email") or decoded.get("uid"), + "eval_transcript_at": datetime.now(timezone.utc).isoformat(), + }) + return {"ok": True, "call_id": call_id} diff --git a/drb-c2-core/tests/test_eval_transcript.py b/drb-c2-core/tests/test_eval_transcript.py new file mode 100644 index 0000000..d60587a --- /dev/null +++ b/drb-c2-core/tests/test_eval_transcript.py @@ -0,0 +1,145 @@ +""" +server-26#163 — the STT eval harness: word_error_rate() and the three routes +that back the /admin "STT Eval" tab. + +Load-bearing property, checked directly: eval annotation must never touch +`transcript`/`transcript_corrected`, re-run extraction, or unlink incidents — +that's PATCH /{call_id}/transcript's job, a production correction with real +side effects. This is pure measurement and must stay pure. +""" +from unittest.mock import AsyncMock, patch + +import pytest +from fastapi.testclient import TestClient + +from app.internal.wer import word_error_rate +from app.main import app +from app.internal.auth import require_admin_token, require_service_or_firebase_token +from app.routers import calls + +client = TestClient(app) + +ADMIN = {"role": "admin", "org_id": "org-A"} + + +def _override(decoded: dict): + # calls.router carries its own router-level require_service_or_firebase_token + # (app/main.py) ON TOP OF each admin route's own require_admin_token — both + # have to be overridden or the router-level one 401s before the route's own + # dependency is ever evaluated. + app.dependency_overrides[require_admin_token] = lambda: decoded + app.dependency_overrides[require_service_or_firebase_token] = lambda: decoded + + +def teardown_function(): + app.dependency_overrides.pop(require_admin_token, None) + app.dependency_overrides.pop(require_service_or_firebase_token, None) + + +# ── word_error_rate ───────────────────────────────────────────────────────── + +def test_identical_transcripts_are_zero_wer(): + assert word_error_rate("K on the 600, I'm on Jackson Avenue.", + "K on the 600, I'm on Jackson Avenue.") == 0.0 + + +def test_case_and_punctuation_are_ignored(): + assert word_error_rate("Home Street and Forest Ave!", "home street and forest ave") == 0.0 + + +def test_one_substitution_out_of_three_words(): + assert word_error_rate("the cat sat", "the cat sit") == pytest.approx(1 / 3) + + +def test_empty_reference_is_undefined_not_zero(): + """A call nobody transcribed must not score as a perfect match.""" + assert word_error_rate("", "anything") is None + assert word_error_rate(None, "anything") is None + + +def test_empty_hypothesis_against_real_reference_is_total_loss(): + assert word_error_rate("home street and forest ave", "") == 1.0 + + +def test_insertion_counts_against_the_hypothesis(): + # reference 3 words, hypothesis adds 2 extra -> 2 insertions / 3 ref words + assert word_error_rate("show me clear", "show me clear right now") == pytest.approx(2 / 3) + + +# ── GET /calls/eval-queue ─────────────────────────────────────────────────── + +def _call(call_id, transcript="a real transcript here", corrected=None, eval_transcript=None, org_id="org-A"): + return { + "call_id": call_id, "org_id": org_id, "started_at": "2026-09-21T00:00:00+00:00", + "transcript": transcript, "transcript_corrected": corrected, "eval_transcript": eval_transcript, + } + + +def test_eval_queue_skips_already_evaluated_and_transcript_less_calls(): + rows = [ + _call("c1", eval_transcript="already done"), + _call("c2", transcript=None), + _call("c3"), + ] + _override(ADMIN) + with patch.object(calls.fstore, "collection_where", AsyncMock(return_value=rows)): + resp = client.get("/calls/eval-queue") + assert resp.status_code == 200 + body = resp.json() + assert [c["call_id"] for c in body["calls"]] == ["c3"] + assert body["matched"] == 1 + + +def test_eval_queue_requires_an_org_scope(): + _override({"role": "admin"}) # platform admin, no org claim + resp = client.get("/calls/eval-queue") + assert resp.status_code == 403 + + +# ── GET /calls/eval-stats ─────────────────────────────────────────────────── + +def test_eval_stats_averages_wer_across_evaluated_calls_only(): + rows = [ + _call("c1", transcript="the cat sat", corrected="the cat sat", eval_transcript="the cat sat"), # 0.0 / 0.0 + _call("c2", transcript="the cat sit", corrected="the cat sat", eval_transcript="the cat sat"), # raw 1/3, corrected 0.0 + _call("c3", eval_transcript=None), # excluded entirely + ] + _override(ADMIN) + with patch.object(calls.fstore, "collection_list", AsyncMock(return_value=rows)): + resp = client.get("/calls/eval-stats") + assert resp.status_code == 200 + body = resp.json() + assert body["eval_count"] == 2 + assert body["raw_wer"] == pytest.approx((0.0 + 1 / 3) / 2, abs=1e-4) + assert body["corrected_wer"] == 0.0 + + +def test_eval_stats_with_nothing_evaluated_yet_reports_none_not_zero(): + _override(ADMIN) + with patch.object(calls.fstore, "collection_list", AsyncMock(return_value=[_call("c1")])): + resp = client.get("/calls/eval-stats") + body = resp.json() + assert body == {"eval_count": 0, "raw_wer": None, "corrected_wer": None} + + +# ── PUT /{call_id}/eval-transcript ────────────────────────────────────────── + +def test_put_eval_transcript_writes_only_eval_fields(): + _override(ADMIN) + existing = _call("c1", transcript="raw text", corrected="corrected text") + with patch.object(calls.fstore, "doc_get", AsyncMock(return_value=existing)), \ + patch.object(calls.fstore, "doc_set", AsyncMock()) as mock_set: + resp = client.put("/calls/c1/eval-transcript", json={"text": "the verified ground truth"}) + assert resp.status_code == 200 + (collection, doc_id, doc), _ = mock_set.await_args + assert collection == "calls" and doc_id == "c1" + assert doc["eval_transcript"] == "the verified ground truth" + assert doc["eval_transcript_at"] + assert "transcript" not in doc and "transcript_corrected" not in doc + + +def test_put_eval_transcript_404s_on_missing_call(): + _override(ADMIN) + with patch.object(calls.fstore, "doc_get", AsyncMock(return_value=None)): + resp = client.put("/calls/nope/eval-transcript", json={"text": "x"}) + assert resp.status_code == 404 diff --git a/drb-frontend/app/admin/page.tsx b/drb-frontend/app/admin/page.tsx index f2de6ba..35d132b 100644 --- a/drb-frontend/app/admin/page.tsx +++ b/drb-frontend/app/admin/page.tsx @@ -4,7 +4,7 @@ import { useAuth } from "@/components/AuthProvider"; import { c2api } from "@/lib/c2api"; import { useEffect, useState, useRef, useCallback } from "react"; import { useRouter } from "next/navigation"; -import type { UserRecord, AuditEntry, UserRole } from "@/lib/types"; +import type { UserRecord, AuditEntry, UserRole, CallRecord } from "@/lib/types"; // --------------------------------------------------------------------------- // Shared primitives @@ -1047,16 +1047,193 @@ function StaleCallsTab() { ); } +// --------------------------------------------------------------------------- +// STT eval (server-26#163) — the eval harness for real transcription +// accuracy. Separate from patchTranscript's "fix this call" flow: this never +// re-runs extraction or touches an incident, it only records what was +// actually said next to what Whisper heard, so eval-stats can report a real +// WER instead of a guess. Built to be worked in short sessions, a handful of +// calls at a time, over however many sittings it takes — not a one-shot form. +// --------------------------------------------------------------------------- + +function fmtPct(x: number | null | undefined): string { + return x === null || x === undefined ? "—" : `${(x * 100).toFixed(1)}%`; +} + +function EvalStatsBar({ stats }: { stats: { eval_count: number; raw_wer: number | null; corrected_wer: number | null } | null }) { + return ( +
+
+

Calls verified

+

{stats?.eval_count ?? "—"}

+
+
+

Raw WER (whisper-1)

+

{fmtPct(stats?.raw_wer)}

+
+
+

Corrected WER (shipped)

+

{fmtPct(stats?.corrected_wer)}

+
+ {stats && stats.eval_count > 0 && stats.eval_count < 20 && ( +

+ fewer than 20 calls — numbers will move a lot until this grows +

+ )} +
+ ); +} + +function SttEvalTab() { + const [stats, setStats] = useState<{ eval_count: number; raw_wer: number | null; corrected_wer: number | null } | null>(null); + const [queue, setQueue] = useState([]); + const [cursor, setCursor] = useState(null); + const [exhausted, setExhausted] = useState(false); + const [draft, setDraft] = useState(""); + const [loadingBatch, setLoadingBatch] = useState(false); + const [saving, setSaving] = useState(false); + const [error, setError] = useState(null); + const fetching = useRef(false); + + const current = queue[0] ?? null; + + const refreshStats = useCallback(() => { + c2api.getEvalStats().then(setStats).catch(() => { /* stats are a nice-to-have, not load-bearing */ }); + }, []); + + const loadBatch = useCallback(async () => { + if (fetching.current) return; + fetching.current = true; + setLoadingBatch(true); + setError(null); + try { + const res = await c2api.getEvalQueue(5, cursor); + setQueue((q) => [...q, ...res.calls]); + setCursor(res.next_cursor); + if (res.calls.length === 0 && !res.next_cursor) setExhausted(true); + } catch (e) { + setError(String(e)); + } finally { + setLoadingBatch(false); + fetching.current = false; + } + }, [cursor]); + + useEffect(() => { refreshStats(); }, [refreshStats]); + + // Auto-refill: whenever the local queue runs dry and there's more to scan + // (or we haven't checked yet), pull another batch. Covers the sparse-window + // case too — a page with matches:0 but a next_cursor just means "keep + // scanning", not "done", so this fires again on its own. + useEffect(() => { + if (queue.length === 0 && !exhausted) loadBatch(); + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [queue.length, exhausted]); + + useEffect(() => { + setDraft(current ? (current.transcript_corrected || current.transcript || "") : ""); + }, [current]); + + async function saveAndNext() { + if (!current) return; + setSaving(true); + setError(null); + try { + await c2api.putEvalTranscript(current.call_id, draft); + setQueue((q) => q.slice(1)); + refreshStats(); + } catch (e) { + setError(String(e)); + } finally { + setSaving(false); + } + } + + function skip() { + setQueue((q) => q.slice(1)); + } + + return ( +
+

+ Listen to the audio, correct the transcript below until it matches what was actually said, then save. + This never touches the call's real transcript or re-runs anything — it only records ground truth + for measuring the pipeline. Do as many or as few as you have time for; it picks up where you left off. +

+ + + + {error && ( +
+

{error}

+
+ )} + + {current ? ( +
+
+ {new Date(current.started_at).toLocaleString()} + {current.talkgroup_name || (current.talkgroup_id ? `TGID ${current.talkgroup_id}` : "unknown talkgroup")} +
+ + {current.audio_url ? ( + // eslint-disable-next-line jsx-a11y/media-has-caption +