incidents: severity-scaled quiet timer, reopen-on-link, thin calls don't fill the cap
Hand-labelling the 09-22 10:00-12:00 ET replay window (server-26#170, answer key replay_groundtruth_0922.json) found ~25 real incidents, of which only ~5 had an audible clear — most jobs clear by MDT, so the quiet timer is the close for most incidents and a flat 90 minutes left a lockout or a plate check "active" on the portal an hour after it ended. - summarizer: timer close after 30 min quiet for routine/minor, 60 moderate, 90 major/unknown. A timer close is provisional: reopenable=True. - correlator: reopenable incidents inside incident_reopen_window_minutes (90, since last substantive call) stay candidates; linking a call to one reopens it (status active, reopened_count++). The sweep expires the flag so the reopenable pool stays bounded. Real clears (units_cleared, llm_closure) are never reopenable. - cap: incident_max_calls counts substantive calls only (substantive_call_count). The bridge MVA hit 40 in 32 min with ~40% thin replies, split in half, and the second half took another job's title. c2-core: 467 pass. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
731b54bed9
commit
969d175a67
@@ -618,7 +618,14 @@ def _incident_at_capacity(inc: dict, now: datetime) -> Optional[str]:
|
||||
it has and still auto-resolves on the normal idle sweep. It just stops
|
||||
being a candidate, so the next call opens a fresh incident.
|
||||
"""
|
||||
call_count = len(inc.get("call_ids") or [])
|
||||
# Content-free replies ("10-4", "Cut.", "Copy") are what filled the cap:
|
||||
# the 09-22 bridge MVA hit 40 calls in 32 minutes with ~40% of them thin,
|
||||
# split in half, and its second half took a different job's title. Only
|
||||
# substantive calls count; incidents written before this field existed
|
||||
# fall back to the raw count.
|
||||
call_count = inc.get("substantive_call_count")
|
||||
if call_count is None:
|
||||
call_count = len(inc.get("call_ids") or [])
|
||||
if call_count >= settings.incident_max_calls:
|
||||
return f"call_cap:{call_count}"
|
||||
span = _incident_span_minutes(inc, now)
|
||||
@@ -851,8 +858,16 @@ async def _build_context(
|
||||
# the whole collection rather than being unable to correlate at all.
|
||||
if org_id is not None:
|
||||
all_active = await fstore.collection_list("incidents", status="active", org_id=org_id)
|
||||
reopenable = await fstore.collection_list("incidents", status="resolved", reopenable=True, org_id=org_id)
|
||||
else:
|
||||
all_active = await fstore.collection_list("incidents", status="active")
|
||||
reopenable = await fstore.collection_list("incidents", status="resolved", reopenable=True)
|
||||
# A timer-closed incident is provisional (summarizer._resolve_stale_incidents):
|
||||
# inside its reopen window it is still a candidate, and linking a call to
|
||||
# it reopens it (_update_incident). The fast path's own recency gate
|
||||
# (tg_fast_path_idle_minutes) still applies to it like any other candidate.
|
||||
reopen_window = timedelta(minutes=settings.incident_reopen_window_minutes)
|
||||
all_active += [inc for inc in reopenable if _idle_gate_minutes(inc, now) <= reopen_window.total_seconds() / 60]
|
||||
# Incidents past the hard caps are removed from the candidate pool here, so
|
||||
# neither the rules engine nor the LLM tier (which reads ctx["recent"] /
|
||||
# ctx["all_active"]) can propose linking into one.
|
||||
@@ -2086,6 +2101,10 @@ async def _update_incident(
|
||||
# thin traffic rides along without extending its life.
|
||||
if refresh_activity:
|
||||
updates["updated_at"] = _floor_at_started_at(inc, now).isoformat()
|
||||
updates["substantive_call_count"] = (
|
||||
inc.get("substantive_call_count")
|
||||
if inc.get("substantive_call_count") is not None else len(inc.get("call_ids") or [])
|
||||
) + 1
|
||||
else:
|
||||
updates["last_thin_at"] = now.isoformat()
|
||||
# Update incident type when a re-classified call provides a concrete type.
|
||||
@@ -2103,6 +2122,12 @@ async def _update_incident(
|
||||
# Signal-based auto-resolve: every tracked unit has cleared, none still active.
|
||||
# Requires at least one unit to have explicitly signalled back-in-service so we
|
||||
# don't fire on incidents where units were never tracked (no unit mentions at all).
|
||||
if inc.get("status") == "resolved":
|
||||
# A timer close was provisional and a related call just arrived.
|
||||
updates.update({"status": "active", "resolved_at": None, "resolved_via": None,
|
||||
"reopenable": False, "reopened_count": (inc.get("reopened_count") or 0) + 1})
|
||||
logger.info(f"Correlator: reopened timer-closed incident {incident_id} (call {call_id})")
|
||||
|
||||
if units_cleared and not units_active:
|
||||
updates["status"] = "resolved"
|
||||
updates["resolved_at"] = now.isoformat()
|
||||
@@ -2171,6 +2196,7 @@ async def _create_incident(
|
||||
**location_fields,
|
||||
"location_mentions": [location] if location else [],
|
||||
"call_ids": [call_id],
|
||||
"substantive_call_count": 1,
|
||||
"talkgroup_ids": [str(talkgroup_id)] if talkgroup_id is not None else [],
|
||||
"system_ids": [system_id] if system_id else [],
|
||||
"tags": tags + ["auto-generated"],
|
||||
|
||||
@@ -142,15 +142,41 @@ async def _summarize_incident(inc: dict) -> None:
|
||||
await fstore.doc_set("incidents", incident_id, updates)
|
||||
|
||||
|
||||
def _auto_resolve_minutes(inc: dict) -> int:
|
||||
"""Quiet time before a timer close, by severity (see config: incident_auto_resolve_minutes_*)."""
|
||||
sev = (inc.get("severity") or "").lower()
|
||||
if sev in ("routine", "minor"):
|
||||
return settings.incident_auto_resolve_minutes_routine
|
||||
if sev == "moderate":
|
||||
return settings.incident_auto_resolve_minutes_moderate
|
||||
return settings.incident_auto_resolve_minutes
|
||||
|
||||
|
||||
async def _expire_reopen_windows(now) -> None:
|
||||
"""A timer-closed incident stops being reopenable once its window passes,
|
||||
so the correlator's reopenable pool stays bounded."""
|
||||
window = timedelta(minutes=settings.incident_reopen_window_minutes)
|
||||
for inc in await fstore.collection_list("incidents", status="resolved", reopenable=True):
|
||||
try:
|
||||
updated = datetime.fromisoformat(str(inc.get("updated_at", "")).replace("Z", "+00:00"))
|
||||
if updated.tzinfo is None:
|
||||
updated = updated.replace(tzinfo=timezone.utc)
|
||||
except ValueError:
|
||||
updated = None
|
||||
if updated is None or now - updated > window:
|
||||
await fstore.doc_set("incidents", inc["incident_id"], {"reopenable": False})
|
||||
|
||||
|
||||
async def _resolve_stale_incidents() -> None:
|
||||
"""Auto-resolve active incidents that have had no new calls for incident_auto_resolve_minutes."""
|
||||
"""Timer-close active incidents that have been quiet longer than their severity allows."""
|
||||
from app.internal import clock
|
||||
await _expire_reopen_windows(clock.now())
|
||||
all_active = await fstore.collection_list("incidents", status="active")
|
||||
if not all_active:
|
||||
return
|
||||
|
||||
from app.internal import clock
|
||||
now = clock.now()
|
||||
cutoff = timedelta(minutes=settings.incident_auto_resolve_minutes)
|
||||
count = 0
|
||||
|
||||
for inc in all_active:
|
||||
@@ -164,11 +190,12 @@ async def _resolve_stale_incidents() -> None:
|
||||
if updated_dt.tzinfo is None:
|
||||
updated_dt = updated_dt.replace(tzinfo=timezone.utc)
|
||||
idle_minutes = (now - updated_dt).total_seconds() / 60
|
||||
if idle_minutes > settings.incident_auto_resolve_minutes:
|
||||
if idle_minutes > _auto_resolve_minutes(inc):
|
||||
await fstore.doc_set("incidents", incident_id, {
|
||||
"status": "resolved",
|
||||
"resolved_at": now.isoformat(),
|
||||
"resolved_via": "idle_timeout",
|
||||
"reopenable": True,
|
||||
})
|
||||
from app.internal.incident_correlator import maybe_resolve_parent
|
||||
await maybe_resolve_parent(incident_id)
|
||||
|
||||
Reference in New Issue
Block a user