config/ai_features was not the switch it was documented to be. Three paths spent money with it off, and one path read it wrong, so per-system opt-outs did not opt anything out. - Correlation in the ingest pipeline tested the raw global flag instead of the per-system resolution. With a system opted out, extraction was skipped but the no-scenes fallback still correlated the call with empty tags, taking the thin/recency path and attaching it to whatever incident was most recent on that system. The opt-out did not disable correlation, it disabled good correlation and left the worst kind running. (#75) - Transcript correction ran on every transcribed call gated only by an env var, spending Gemini tokens and a Places lookup per proposed location. An "STT-only" window was never STT-only and its cost could not be attributed. Now behind transcript_correction_enabled. (#76) - _run_extraction_pipeline and the vocabulary learner, both reachable from PATCH /calls/{id}/transcript, checked no flags at all. (#76, #81) The flag resolver now lives in feature_flags.resolve_flags() rather than as a local helper in upload.py. Three copies of that logic is how #75 happened. PATCH /calls/{id}/transcript now refuses with 409 when correlation is off. That route wipes tags, severity, location, units, embedding and unlinks the call from every incident before queueing re-extraction. Gating extraction alone would have made it destructive-only in the standing flags-off configuration: the call left blank and orphaned forever, with the route still answering 200. The wipe and the rebuild are one transaction in intent, so it refuses before the first write. Also: the summarizer's stale-incident sweep is no longer behind summaries_enabled. It is pure Firestore with no model call in it, and gating it meant nothing auto-resolved while AI was off - so every incident stayed active forever and the candidate set every correlation reads kept growing. transcript_correction_enabled is documented as NOT a pure cost lever. The corrector is also the noise gate that sets not_speech; with it off, recogniser noise reaches extraction as a real transcript, comes back thin, and auto-attaches. Never open an evaluation window with correction off and correlation on. 14 tests added covering flag precedence, both pipeline paths, the 409, the correction gate and the summarizer no-op. Suite: 264 passed. Refs #75, #76, #81, #45.
383 lines
16 KiB
Python
383 lines
16 KiB
Python
import secrets
|
|
from typing import Optional
|
|
from datetime import datetime, timezone
|
|
from fastapi import APIRouter, BackgroundTasks, UploadFile, File, Form, HTTPException, Security
|
|
from fastapi.security import HTTPBearer, HTTPAuthorizationCredentials
|
|
from app.internal.storage import upload_audio
|
|
from app.internal import dedup
|
|
from app.internal import firestore as fstore
|
|
from app.internal.logger import logger
|
|
from app.config import settings
|
|
|
|
router = APIRouter(tags=["upload"])
|
|
|
|
_bearer = HTTPBearer(auto_error=False)
|
|
|
|
|
|
@router.post("/upload")
|
|
async def upload_call_audio(
|
|
background_tasks: BackgroundTasks,
|
|
file: UploadFile = File(...),
|
|
call_id: str = Form(...),
|
|
node_id: str = Form(...),
|
|
talkgroup_id: Optional[int] = Form(None),
|
|
talkgroup_name: Optional[str] = Form(None),
|
|
system_id: Optional[str] = Form(None),
|
|
credentials: Optional[HTTPAuthorizationCredentials] = Security(_bearer),
|
|
):
|
|
"""
|
|
Receive an audio recording from an edge node.
|
|
Upload to GCS, update the call document in Firestore with the audio URL,
|
|
then kick off the intelligence pipeline as a background task.
|
|
"""
|
|
# Verify the per-node API key
|
|
if not credentials:
|
|
raise HTTPException(401, "Missing authorization")
|
|
key_doc = await fstore.doc_get("node_keys", node_id)
|
|
if not key_doc:
|
|
logger.warning(f"Upload 401: no key_doc in Firestore for node_id={node_id!r}")
|
|
raise HTTPException(401, "Invalid node API key")
|
|
# compare_digest, not !=, so the comparison cost does not depend on how many
|
|
# leading characters matched. enrollment.py and dynsec.py were explicit about
|
|
# this for the same class of credential; this route was the odd one out.
|
|
stored_key = key_doc.get("api_key") or ""
|
|
if not secrets.compare_digest(stored_key, credentials.credentials):
|
|
logger.warning(
|
|
f"Upload 401: key mismatch for node_id={node_id!r} "
|
|
f"(received prefix: {credentials.credentials[:8]}...)"
|
|
)
|
|
raise HTTPException(401, "Invalid node API key")
|
|
|
|
data = await file.read()
|
|
if not data:
|
|
raise HTTPException(400, "Empty file.")
|
|
if len(data) > settings.upload_max_bytes:
|
|
raise HTTPException(413, f"File too large (max {settings.upload_max_bytes // (1024*1024)} MB).")
|
|
|
|
gcs_uri = await upload_audio(data, file.filename or "", call_id=call_id)
|
|
|
|
if gcs_uri:
|
|
try:
|
|
# Canonical object location only. The playback link is minted per
|
|
# read in storage.playback_url() — nothing durable is stored here.
|
|
# org_id is stamped defensively here too (not just in
|
|
# mqtt_handler.py's call_start/call_end): key_doc above proves this
|
|
# node_id is real and authenticated, so resolving org_id from the
|
|
# node doc here covers a call whose Firestore doc was somehow
|
|
# never written by call_start (the upload is otherwise the
|
|
# authoritative record of which node this audio came from).
|
|
node = await fstore.doc_get_cached("nodes", node_id)
|
|
updates = {"audio_gcs_uri": gcs_uri}
|
|
if node and node.get("org_id"):
|
|
updates["org_id"] = node["org_id"]
|
|
await fstore.doc_set("calls", call_id, updates)
|
|
except Exception as e:
|
|
logger.warning(f"Could not update call {call_id} with audio_gcs_uri: {e}")
|
|
|
|
# Another node in range recorded the same transmission. Keep the audio
|
|
# (it may be the cleaner capture) but don't transcribe or correlate it
|
|
# a second time — see app/internal/dedup.py.
|
|
call_doc = await fstore.doc_get("calls", call_id)
|
|
duplicate_of = await dedup.find_duplicate_of(call_doc) if call_doc else None
|
|
if duplicate_of:
|
|
await fstore.doc_set("calls", call_id, {"duplicate_of": duplicate_of})
|
|
logger.info(
|
|
f"Call {call_id} from {node_id} duplicates {duplicate_of} "
|
|
f"— audio kept, AI pipeline skipped."
|
|
)
|
|
return {"url": gcs_uri, "duplicate_of": duplicate_of}
|
|
|
|
background_tasks.add_task(
|
|
_run_intelligence_pipeline,
|
|
call_id=call_id,
|
|
node_id=node_id,
|
|
system_id=system_id,
|
|
talkgroup_id=talkgroup_id,
|
|
talkgroup_name=talkgroup_name,
|
|
gcs_uri=gcs_uri,
|
|
)
|
|
|
|
return {"url": gcs_uri}
|
|
|
|
|
|
async def _correlate_with_consensus(
|
|
call_id: str,
|
|
node_id: str,
|
|
system_id: Optional[str],
|
|
talkgroup_id: Optional[int],
|
|
talkgroup_name: Optional[str],
|
|
tags: list[str],
|
|
incident_type: Optional[str],
|
|
location: Optional[str],
|
|
location_coords: Optional[dict],
|
|
units: Optional[list] = None,
|
|
vehicles: Optional[list] = None,
|
|
cleared_units: Optional[list] = None,
|
|
reassignment: bool = False,
|
|
) -> Optional[str]:
|
|
"""
|
|
Consensus correlator: runs the rules engine and the cheap LLM in sequence.
|
|
If they agree the rules decision is committed directly.
|
|
If they disagree a smarter tiebreaker LLM makes the final call.
|
|
|
|
Falls back to rules-only when GEMINI_API_KEY is absent, the call is
|
|
content-free (thin), or any LLM call fails.
|
|
"""
|
|
from app.internal import incident_correlator, llm_correlator
|
|
|
|
preview = await incident_correlator.preview_correlation(
|
|
call_id=call_id, node_id=node_id, system_id=system_id,
|
|
talkgroup_id=talkgroup_id, talkgroup_name=talkgroup_name,
|
|
tags=tags, incident_type=incident_type, location=location,
|
|
location_coords=location_coords, units=units, vehicles=vehicles,
|
|
cleared_units=cleared_units, reassignment=reassignment,
|
|
)
|
|
ctx = preview["ctx"]
|
|
rules_decision = preview["decision"]
|
|
|
|
llm_decision = await llm_correlator.decide(call_id, ctx)
|
|
|
|
if llm_decision is None:
|
|
# LLM unavailable, skipped (thin call), or errored — rules wins.
|
|
rules_decision["corr_debug"]["corr_consensus"] = "rules_only"
|
|
return await incident_correlator.apply_correlation(preview)
|
|
|
|
if llm_correlator.decisions_agree(rules_decision, llm_decision):
|
|
rules_decision["corr_debug"]["corr_consensus"] = "agreed"
|
|
rules_decision["corr_debug"]["corr_llm_reasoning"] = llm_decision.get("reasoning", "")
|
|
return await incident_correlator.apply_correlation(preview)
|
|
|
|
# Disagree — escalate to the smarter tiebreaker.
|
|
logger.info(
|
|
f"Consensus disagreement for call {call_id}: "
|
|
f"rules={rules_decision['action']} vs llm={llm_decision['action']} — tiebreak"
|
|
)
|
|
final = await llm_correlator.tiebreak(rules_decision, llm_decision, ctx)
|
|
final["corr_debug"]["corr_consensus"] = "tiebreak"
|
|
final["corr_debug"]["corr_rules_action"] = rules_decision["action"]
|
|
final["corr_debug"]["corr_llm_action"] = llm_decision["action"]
|
|
return await incident_correlator.apply_correlation({"decision": final, "ctx": ctx})
|
|
|
|
|
|
async def _resolve_flags(system_id: Optional[str]):
|
|
"""
|
|
Resolve AI feature flags for a given system.
|
|
|
|
Thin alias for `feature_flags.resolve_flags` — the resolver lives there
|
|
because transcription and the calls router need the same answer, and three
|
|
copies of it is how server-26#75 happened in the first place.
|
|
"""
|
|
from app.internal.feature_flags import resolve_flags
|
|
|
|
return await resolve_flags(system_id)
|
|
|
|
|
|
async def _run_extraction_pipeline(
|
|
call_id: str,
|
|
node_id: str,
|
|
system_id: Optional[str],
|
|
talkgroup_id: Optional[int],
|
|
talkgroup_name: Optional[str],
|
|
transcript: str,
|
|
segments: Optional[list] = None,
|
|
preserve_transcript_correction: bool = False,
|
|
) -> None:
|
|
"""Run steps 2-4 of the intelligence pipeline using an existing transcript."""
|
|
from app.internal import intelligence, incident_correlator, alerter
|
|
|
|
flags, _flag = await _resolve_flags(system_id)
|
|
|
|
incident_ids: list[str] = []
|
|
all_tags: list[str] = []
|
|
|
|
if _flag("correlation_enabled"):
|
|
# Step 2: Scene detection + intelligence extraction.
|
|
# Returns one scene per distinct incident detected in the recording.
|
|
scenes = await intelligence.extract_scenes(
|
|
call_id, transcript, talkgroup_name,
|
|
talkgroup_id=talkgroup_id, system_id=system_id, segments=segments,
|
|
node_id=node_id,
|
|
preserve_transcript_correction=preserve_transcript_correction,
|
|
)
|
|
|
|
# Step 3: Correlate each scene to an incident independently.
|
|
for scene in scenes:
|
|
all_tags.extend(scene["tags"])
|
|
# When dispatch is pulling a unit to a NEW call (reassignment), suppress unit
|
|
# overlap so the new scene doesn't chain into the unit's previous incident.
|
|
is_reassignment = bool(scene.get("reassignment"))
|
|
corr_units = [] if is_reassignment else scene.get("units")
|
|
incident_id = await _correlate_with_consensus(
|
|
call_id=call_id,
|
|
node_id=node_id,
|
|
system_id=system_id,
|
|
talkgroup_id=talkgroup_id,
|
|
talkgroup_name=talkgroup_name,
|
|
tags=scene["tags"],
|
|
incident_type=scene["incident_type"],
|
|
location=scene["location"],
|
|
location_coords=scene["location_coords"],
|
|
units=corr_units,
|
|
vehicles=scene.get("vehicles"),
|
|
cleared_units=scene.get("cleared_units"),
|
|
reassignment=is_reassignment,
|
|
)
|
|
if incident_id and incident_id not in incident_ids:
|
|
incident_ids.append(incident_id)
|
|
if scene["resolved"] and incident_id:
|
|
await fstore.doc_set("incidents", incident_id, {
|
|
"status": "resolved",
|
|
"resolved_at": datetime.now(timezone.utc).isoformat(),
|
|
})
|
|
await incident_correlator.maybe_resolve_parent(incident_id)
|
|
logger.info(f"Auto-resolved incident {incident_id} (LLM closure detection)")
|
|
else:
|
|
scope = "globally" if not flags["correlation_enabled"] else f"system {system_id}"
|
|
logger.info(f"Correlation disabled ({scope}) — skipping scene extraction and correlation for call {call_id} (reprocess)")
|
|
|
|
if incident_ids:
|
|
await fstore.doc_set("calls", call_id, {"incident_ids": incident_ids})
|
|
|
|
# Step 4: Alert dispatch — run once with merged tags from all scenes.
|
|
await alerter.check_and_dispatch(
|
|
call_id=call_id,
|
|
node_id=node_id,
|
|
talkgroup_id=talkgroup_id,
|
|
talkgroup_name=talkgroup_name,
|
|
tags=list(dict.fromkeys(all_tags)),
|
|
transcript=transcript,
|
|
)
|
|
|
|
|
|
async def _run_intelligence_pipeline(
|
|
call_id: str,
|
|
node_id: str,
|
|
system_id: Optional[str],
|
|
talkgroup_id: Optional[int],
|
|
talkgroup_name: Optional[str],
|
|
gcs_uri: Optional[str],
|
|
) -> None:
|
|
"""
|
|
Post-upload intelligence pipeline (runs as a background task):
|
|
1. Transcribe audio via Google STT
|
|
2. Detect scenes + extract intelligence (one result per incident in recording)
|
|
3. Correlate each scene with existing incidents (or create new ones)
|
|
4. Check alert rules and dispatch notifications
|
|
"""
|
|
from app.internal import transcription, intelligence, incident_correlator, alerter, talkgroups
|
|
|
|
# The node only sends talkgroup_name when OP25 had it in the loaded tags
|
|
# file, so it arrives empty for exactly the talkgroups C2 can name from the
|
|
# system config. Resolve it once, here, at the single funnel both /upload
|
|
# and /calls/{id}/reprocess pass through — everything downstream (the
|
|
# dispatch-channel test, scene extraction, and the incident title) then
|
|
# gets a real name instead of "TGID 9048". server-26#34.
|
|
_call_doc = await fstore.doc_get("calls", call_id)
|
|
talkgroup_name = await talkgroups.resolve(
|
|
system_id, talkgroup_id, hint=talkgroup_name, call_doc=_call_doc,
|
|
)
|
|
# Backfill the call document too, so the archive and the orphan panel stop
|
|
# showing a bare TGID for a channel we can now name.
|
|
if talkgroup_name and _call_doc is not None and not _call_doc.get("talkgroup_name"):
|
|
try:
|
|
await fstore.doc_set("calls", call_id, {"talkgroup_name": talkgroup_name})
|
|
except Exception as e:
|
|
logger.warning(f"Could not backfill talkgroup_name on call {call_id}: {e}")
|
|
|
|
flags, _flag = await _resolve_flags(system_id)
|
|
|
|
transcript: Optional[str] = None
|
|
segments: list[dict] = []
|
|
|
|
# Step 1: Transcription
|
|
if gcs_uri:
|
|
if _flag("stt_enabled"):
|
|
transcript, segments = await transcription.transcribe_call(
|
|
call_id, gcs_uri, talkgroup_name,
|
|
system_id=system_id, talkgroup_id=talkgroup_id,
|
|
)
|
|
else:
|
|
scope = "globally" if not flags["stt_enabled"] else f"system {system_id}"
|
|
logger.info(f"STT disabled ({scope}) — skipping transcription for call {call_id}")
|
|
|
|
# Step 2: Scene detection + intelligence extraction
|
|
scenes: list[dict] = []
|
|
if _flag("correlation_enabled"):
|
|
if transcript:
|
|
scenes = await intelligence.extract_scenes(
|
|
call_id, transcript, talkgroup_name,
|
|
talkgroup_id=talkgroup_id, system_id=system_id, segments=segments,
|
|
node_id=node_id,
|
|
)
|
|
else:
|
|
scope = "globally" if not flags["correlation_enabled"] else f"system {system_id}"
|
|
logger.info(f"Correlation disabled ({scope}) — skipping scene extraction and correlation for call {call_id}")
|
|
|
|
# Step 3: Correlate each scene independently.
|
|
# A single recording can produce multiple incidents on a busy channel.
|
|
incident_ids: list[str] = []
|
|
all_tags: list[str] = []
|
|
if _flag("correlation_enabled"):
|
|
for scene in scenes:
|
|
all_tags.extend(scene["tags"])
|
|
is_reassignment = bool(scene.get("reassignment"))
|
|
corr_units = [] if is_reassignment else scene.get("units")
|
|
incident_id = await _correlate_with_consensus(
|
|
call_id=call_id,
|
|
node_id=node_id,
|
|
system_id=system_id,
|
|
talkgroup_id=talkgroup_id,
|
|
talkgroup_name=talkgroup_name,
|
|
tags=scene["tags"],
|
|
incident_type=scene["incident_type"],
|
|
location=scene["location"],
|
|
location_coords=scene["location_coords"],
|
|
units=corr_units,
|
|
vehicles=scene.get("vehicles"),
|
|
cleared_units=scene.get("cleared_units"),
|
|
reassignment=is_reassignment,
|
|
)
|
|
if incident_id and incident_id not in incident_ids:
|
|
incident_ids.append(incident_id)
|
|
if scene["resolved"] and incident_id:
|
|
await fstore.doc_set("incidents", incident_id, {
|
|
"status": "resolved",
|
|
"resolved_at": datetime.now(timezone.utc).isoformat(),
|
|
})
|
|
await incident_correlator.maybe_resolve_parent(incident_id)
|
|
logger.info(f"Auto-resolved incident {incident_id} (LLM closure detection)")
|
|
|
|
# Correlator also runs for calls with no scenes (unclassified) to attempt
|
|
# talkgroup-based linking even when no transcript could be produced.
|
|
# Skip when extraction flagged the call — garbage or too-short transcripts
|
|
# carry no signal and would only attach spuriously via the thin path.
|
|
if not scenes:
|
|
_call_doc = await fstore.doc_get("calls", call_id)
|
|
if not (_call_doc or {}).get("skip_reason"):
|
|
incident_id = await _correlate_with_consensus(
|
|
call_id=call_id,
|
|
node_id=node_id,
|
|
system_id=system_id,
|
|
talkgroup_id=talkgroup_id,
|
|
talkgroup_name=talkgroup_name,
|
|
tags=[],
|
|
incident_type=None,
|
|
location=None,
|
|
location_coords=None,
|
|
)
|
|
if incident_id:
|
|
incident_ids.append(incident_id)
|
|
|
|
if incident_ids:
|
|
await fstore.doc_set("calls", call_id, {"incident_ids": incident_ids})
|
|
|
|
# Step 4: Alert dispatch (always runs — talkgroup ID rules don't need a transcript)
|
|
await alerter.check_and_dispatch(
|
|
call_id=call_id,
|
|
node_id=node_id,
|
|
talkgroup_id=talkgroup_id,
|
|
talkgroup_name=talkgroup_name,
|
|
tags=list(dict.fromkeys(all_tags)),
|
|
transcript=transcript,
|
|
)
|