correlator/intelligence: let 10-8s actually close incidents

Replay of 09-22 10:00-12:00 ET (server-26#170): 0 of 19 incidents resolved
on a clear, 19 on the idle timer, although 25 transmissions said 10-8/clear.
Three independent breaks:

1. Short clears never reached extraction. "45-9, I'm clear." is <=5 words,
   so extract_scenes skipped it before GPT and cleared_units stayed empty.
   A rule parser now names the unit when it precedes the status word
   (never guesses: "10-8, 10-8." / "CMT clear." clear nobody) and returns a
   minimal scene that links by unit overlap but cannot open an incident.
2. Clearance compared unit strings exactly, so "11-Adam" clearing never
   removed "11 Adam". Now by _normalize_unit key.
3. units_active collected "Desk", "Central", "Division", "unknown", plate
   phonetics — none of which ever clear, so all-clear could never pass.
   Only units carrying a number (and not a ten-code) are tracked now; the
   rest stay in `units` for matching.

c2-core: 463 pass.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Logan Cusano
2026-09-26 16:49:30 -04:00
co-authored by Claude Opus 5.5
parent 6e82ee8579
commit 8eac32caf5
3 changed files with 132 additions and 7 deletions
+56 -2
View File
@@ -247,18 +247,26 @@ async def extract_scenes(
f"Intelligence: call {call_id} — transcript too short for extraction "
f"({len(transcript.split())} words), skipping"
)
cleared_unit = _short_clearance_unit(transcript)
try:
# Severity is still recorded: a five-word acknowledgement is genuinely
# routine traffic, and downstream code treats a missing severity as
# "not yet processed" rather than "nothing happened".
await fstore.doc_set("calls", call_id, {
updates = {
"skip_reason": "transcript_too_short",
"severity": "routine",
"chatter_classifier_verdict": chatter_is_chatter,
"chatter_classifier_reason": chatter_reason,
})
}
if cleared_unit:
updates["units"] = [cleared_unit]
updates["cleared_units"] = [cleared_unit]
await fstore.doc_set("calls", call_id, updates)
except Exception:
pass
if cleared_unit:
logger.info(f"Intelligence: call {call_id} — short clearance from {cleared_unit!r}")
return [_clearance_scene(transcript, cleared_unit)]
return []
try:
@@ -469,6 +477,52 @@ async def extract_scenes(
return processed
# "45-9, I'm clear." / "Vehicle 1, clear." / "Car 12 10-8" — a unit reporting
# itself back in service is the one signal that ends an incident, and it is
# almost always five words or fewer, which is exactly the population the
# too-short skip above keeps away from GPT. In the first replay
# (server-26#170, 09-22 10:00-12:00 ET) 25 transmissions said 10-8/clear and
# 2 reached cleared_units. Rule-based on purpose: no model call, and only a
# unit named BEFORE the status word counts, so "10-8, 10-8." or "CMT clear."
# (no number) clears nobody rather than guessing.
_CLEAR_WORD_RE = re.compile(
r"\b(clear|10-?8|10-?98|back in service|in service|available)\b", re.IGNORECASE
)
_TEN_CODE_TOKEN_RE = re.compile(r"^10-?\d{1,2}$")
_UNIT_PREFIX_WORDS = {"unit", "car", "vehicle", "engine", "ladder", "medic", "rescue", "post", "truck", "squad"}
def _short_clearance_unit(transcript: str) -> Optional[str]:
m = _CLEAR_WORD_RE.search(transcript or "")
if not m:
return None
before = [t.strip(".,;:!?") for t in transcript[: m.start()].split()]
before = [t for t in before if t]
for i, tok in enumerate(before[:4]):
if not any(ch.isdigit() for ch in tok) or _TEN_CODE_TOKEN_RE.match(tok):
continue
prev = before[i - 1] if i else ""
if prev.lower() in _UNIT_PREFIX_WORDS:
return f"{prev} {tok}"
nxt = before[i + 1] if i + 1 < len(before) else ""
if nxt.isalpha() and nxt.lower() not in {"i'm", "im", "is", "are", "to", "we're", "copy"} \
and nxt[0].isupper():
return f"{tok} {nxt}" # "11 Adam, clear"
return tok
return None
def _clearance_scene(transcript: str, unit: str) -> dict:
"""A minimal scene for a rule-parsed clearance: the unit, and nothing that
could make the incident-creation gate open a new incident for it."""
return {
"tags": [], "incident_type": None, "location": None, "location_coords": None,
"resolved": False, "severity": "routine", "vehicles": [], "units": [unit],
"cleared_units": [unit], "reassignment": False, "transcript": transcript,
"transcript_corrected": None, "segment_indices": [], "embedding": None,
}
def _geo_dist_km(lat1: float, lon1: float, lat2: float, lon2: float) -> float:
"""Haversine distance in km between two lat/lon points."""
R = 6371.0