Replay: fail fast on a dead AI account; extraction reports to ai_health
First replay (290 calls, 09-22 10:00-12:00 ET) produced 0 incidents and no errors: every gpt-4o-mini extraction failed and _sync_extract swallowed it as "no scenes". Same shape as #169 — and the live extraction tier in /health/ai had no reporter at all, so this has been invisible in production too. - intelligence: API failures propagate out of _sync_extract; extract_scenes reports them to ai_health ("extraction" tier, billing/dead-model classified) and still returns [] so the pipeline degrades as before. - ai_health: inside a replay sandbox, failures go to the run's own sink instead of being dropped. - replay: aborts after 5 permanent failures on a tier, naming the cause; run metrics carry ai_failures; UI shows them. - replay estimate: audio minutes from started_at/ended_at (no duration field exists on call docs). - ReplayTab exposes the loaded run on window.__drbReplay for in-page analysis. c2-core: 458 pass. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
aff3f16d32
commit
ec91a9175f
@@ -39,7 +39,7 @@ from datetime import datetime, timedelta, timezone
|
||||
from typing import Optional
|
||||
|
||||
from app.config import settings
|
||||
from app.internal import clock
|
||||
from app.internal import ai_health, clock
|
||||
from app.internal import firestore as fstore
|
||||
from app.internal.feature_flags import force_flags, unforce_flags
|
||||
from app.internal.logger import logger
|
||||
@@ -174,10 +174,16 @@ def _pipeline_time(call: dict) -> datetime:
|
||||
return _as_dt(call.get("ended_at")) or _call_time(call)
|
||||
|
||||
|
||||
def _duration_s(call: dict) -> float:
|
||||
# Call docs carry no duration field; the node reports start and end.
|
||||
start, end = _as_dt(call.get("started_at")), _as_dt(call.get("ended_at"))
|
||||
return max(0.0, (end - start).total_seconds()) if start and end else 0.0
|
||||
|
||||
|
||||
def estimate(calls: list[dict], mode: str) -> dict:
|
||||
n = len(calls)
|
||||
with_transcript = sum(1 for c in calls if c.get("transcript_corrected") or c.get("transcript"))
|
||||
audio_min = sum(float(c.get("duration_s") or 0) for c in calls) / 60
|
||||
audio_min = sum(_duration_s(c) for c in calls) / 60
|
||||
with_audio = sum(1 for c in calls if c.get("audio_gcs_uri"))
|
||||
# Roughly a third of calls carry a geocodable location (09-22 dump: 92/373).
|
||||
per_call = USD_PER_EXTRACTION + USD_PER_LLM_CORRELATE + USD_PER_GEOCODE / 3
|
||||
@@ -461,6 +467,8 @@ async def _run(run_id: str, org_id: str, calls: list[dict], mode: str,
|
||||
|
||||
sb_token = fstore.enter_sandbox(sandbox_root(run_id))
|
||||
fl_token = force_flags(_flags_for(mode))
|
||||
ai_failures: list = []
|
||||
ai_token = ai_health.collect_sandbox_failures(ai_failures)
|
||||
try:
|
||||
sem = asyncio.Semaphore(PREFETCH)
|
||||
|
||||
@@ -482,6 +490,14 @@ async def _run(run_id: str, org_id: str, calls: list[dict], mode: str,
|
||||
if run_id in _cancel:
|
||||
status = "cancelled"
|
||||
break
|
||||
fatal = _fatal_ai_failure(ai_failures)
|
||||
if fatal:
|
||||
# An unfunded or retired model fails every call the same way;
|
||||
# finishing the run would only produce a sandbox of orphans
|
||||
# that looks like a correlation result and isn't one.
|
||||
status = "failed"
|
||||
errors.append(f"aborted: {fatal}")
|
||||
break
|
||||
|
||||
t = _pipeline_time(call)
|
||||
last_t = t
|
||||
@@ -519,7 +535,7 @@ async def _run(run_id: str, org_id: str, calls: list[dict], mode: str,
|
||||
if prepared["transcript"] and mode != "reuse":
|
||||
progress["extractions"] += 1
|
||||
if mode == "audio":
|
||||
progress["audio_minutes"] += float(call.get("duration_s") or 0) / 60
|
||||
progress["audio_minutes"] += _duration_s(call) / 60
|
||||
except Exception as e:
|
||||
progress["errors"] += 1
|
||||
if len(errors) < 20:
|
||||
@@ -544,12 +560,14 @@ async def _run(run_id: str, org_id: str, calls: list[dict], mode: str,
|
||||
sb_calls = await fstore.collection_list("calls")
|
||||
metrics = compute_metrics(incidents, sb_calls)
|
||||
metrics["est_cost_usd"] = _running_cost(progress, metrics, mode)
|
||||
metrics["ai_failures"] = dict(Counter(f"{f['tier']}: {f['problem']}" for f in ai_failures))
|
||||
except Exception as e:
|
||||
status = "failed"
|
||||
errors.append(f"run: {type(e).__name__}: {e}"[:300])
|
||||
metrics = None
|
||||
logger.error(f"Replay {run_id} failed: {e}")
|
||||
finally:
|
||||
ai_health._sandbox_failures.reset(ai_token)
|
||||
unforce_flags(fl_token)
|
||||
fstore.exit_sandbox(sb_token)
|
||||
_cancel.discard(run_id)
|
||||
@@ -565,6 +583,21 @@ async def _run(run_id: str, org_id: str, calls: list[dict], mode: str,
|
||||
logger.info(f"Replay {run_id} {status}: {progress}")
|
||||
|
||||
|
||||
FATAL_AFTER = 5
|
||||
|
||||
|
||||
def _fatal_ai_failure(failures: list) -> Optional[str]:
|
||||
"""A tier that failed permanently (no credit, dead model) FATAL_AFTER times."""
|
||||
permanent = Counter(
|
||||
f"{f['tier']} ({f['provider']} {f['model']}): {f['problem']}"
|
||||
for f in failures if f.get("permanent")
|
||||
)
|
||||
for what, n in permanent.items():
|
||||
if n >= FATAL_AFTER:
|
||||
return what
|
||||
return None
|
||||
|
||||
|
||||
def _running_cost(progress: dict, metrics: dict, mode: str) -> float:
|
||||
usd = progress["audio_minutes"] * USD_WHISPER_PER_MIN
|
||||
if mode == "audio":
|
||||
|
||||
Reference in New Issue
Block a user