Files
server-26/drb-c2-core/app/config.py
T
Logan Cusano a250c29e3c
Build & Deploy / Build & push images (push) Successful in 4m18s
Build & Deploy / Deploy to VM (push) Successful in 1m35s
Add AI provider degradation registry and alerting (server-26#14)
Three AI dependency failures in one night (retired Gemini model IDs,
depleted Gemini balance, unpayable OpenAI account) each surfaced only
as a single ERROR log line that nobody was watching. Add
app/internal/ai_health.py, a shared in-memory registry that
transcription.py and llm_correlator.py report into on every call
(success and failure), distinguishing permanent conditions (dead
model, dead billing) which alert immediately from transient ones
(rate limits, network blips) which only alert after they persist.
Alerts POST once per degradation episode and once on recovery to an
optional Discord webhook (AI_ALERT_WEBHOOK_URL), reusing alerter.py's
httpx pattern. State is exposed unauthenticated at GET /health/ai
alongside the existing /health.

Closes logan/server-26#14
2026-08-20 03:14:22 -04:00

115 lines
6.0 KiB
Python

from pydantic_settings import BaseSettings
from typing import Optional
class Settings(BaseSettings):
# MQTT
mqtt_broker: str = "localhost"
mqtt_port: int = 1883
mqtt_user: Optional[str] = None
mqtt_pass: Optional[str] = None
# mosquitto's built-in dynamic-security plugin (see app/internal/dynsec.py).
# "admin" is hardcoded by the plugin itself on first boot — not actually
# configurable — kept as a named setting rather than a literal for
# readability. mqtt_dynsec_admin_pass must equal the mosquitto
# container's own MOSQUITTO_DYNSEC_PASSWORD env var (root .env /
# root.env.j2) or c2-core can't administer node credentials at all.
mqtt_dynsec_admin_user: str = "admin"
mqtt_dynsec_admin_pass: Optional[str] = None
# GCP
gcp_credentials_path: Optional[str] = None # None → uses ADC
gcs_bucket: Optional[str] = None # None → audio upload disabled
firestore_database: str = "(default)"
# Node health
node_offline_threshold: int = 90 # seconds without checkin before marking offline
# OpenAI (STT + intelligence)
openai_api_key: Optional[str] = None
stt_model: str = "whisper-1" # whisper-1 | gpt-4o-mini-transcribe | gpt-4o-transcribe
# Google Maps (geocoding)
google_maps_api_key: Optional[str] = None
# Gemini (intelligence extraction, embeddings, incident summaries)
gemini_api_key: Optional[str] = None
# Correlation consensus models
# corr_cheap_model — first-pass LLM correlator (runs on every call)
# corr_smart_model — tiebreaker (only fires when rules and cheap LLM disagree)
# Both IDs below were retired by Google and returned 404 on every call from
# some point before 2026-08-18 until they were corrected. Because a failed
# LLM call falls back to the rules decision, nothing broke loudly -- the
# entire LLM tier and the consensus tiebreak were simply dead in production
# while correlation behaviour was being tuned against rules-only output.
# Verify against https://ai.google.dev/gemini-api/docs/models before changing.
corr_cheap_model: str = "gemini-3.6-flash" # was gemini-2.0-flash (shut down)
corr_smart_model: str = "gemini-2.5-pro" # was gemini-1.5-pro (shut down)
summary_interval_minutes: int = 2 # how often the summary loop runs
correlation_window_hours: int = 2 # slow/location path: max hours since last call
embedding_similarity_threshold: float = 0.93 # slow-path: requires location corroboration
embedding_no_location_threshold: float = 0.97 # slow-path: match without location (very high bar)
embedding_cross_tg_threshold: float = 0.85 # cross-TG path: same dept + 2+ shared units
location_proximity_km: float = 0.5 # radius for location-proximity matching
geocode_max_km: float = 40.0 # reject geocode results farther than this from the node
incident_auto_resolve_minutes: int = 90 # auto-resolve after N minutes with no new calls
unit_continuity_max_idle_minutes: int = 20 # unit-continuity path: skip if incident idle > this
recorrelation_scan_minutes: int = 60 # re-examine orphaned calls ended within this window
tg_fast_path_idle_minutes: int = 90 # fast path: max minutes since incident last updated
# Dispatch channels only: tier-2 thin calls attach to a lone candidate idle < this.
# Was 10, which is long enough for the channel to have moved on to something else:
# on 2026-08-16 a "72 at Holland Station" incident absorbed a Grand Central train
# meet 9.6 min later, and a status check absorbed a records lookup at 9.7 min.
# Across that dump every correct thin attach was <= 3.4 min idle and every wrong
# one was >= 8.2, so 5 separates them with room on both sides. Genuine
# back-and-forth is handled by the 30-second tier-1 path above this.
tg_dispatch_thin_idle_minutes: int = 5
# Vocabulary learning
vocabulary_induction_interval_hours: int = 24 # how often the induction loop runs
vocabulary_induction_sample_tokens: int = 4000 # ~tokens of transcript text sampled per system
# Internal service key — allows server-side services (discord bot) to call C2 without Firebase
service_key: Optional[str] = None
# Fleet-wide token edge nodes present to POST /nodes/enroll on first boot.
# Not a per-node secret — see routers/enrollment.py for why a leaked copy
# of this alone can't steal an already-approved node's key.
enrollment_token: Optional[str] = None
# Upload size limit — reject audio files larger than this (bytes). Default 100 MB.
upload_max_bytes: int = 100 * 1024 * 1024
# Public origin this API is reachable on, e.g. "https://api.drb.example.com".
# Only used to build absolute call-audio playback links: an <audio src> is
# fetched by the browser directly, so a relative path would resolve against
# the frontend origin, not this one.
public_api_url: Optional[str] = None
# How long a minted call-audio playback link stays valid. Long enough for a
# browsing session, short enough that a copied link isn't durable access.
audio_link_ttl_seconds: int = 6 * 60 * 60
# Two nodes hearing the same transmission start recording within about a
# second of each other (measured across node-002/node-PI-2 on TG 9048).
# 10s is generous against clock skew while staying well under the gap
# between genuinely separate transmissions on a busy dispatch channel.
duplicate_window_seconds: int = 10
# CORS — set to your frontend origin(s) in production, e.g. ["https://app.example.com"]
# Defaults to "*" for local development only.
cors_origins: list[str] = ["*"]
# Discord webhook URL that app/internal/ai_health.py posts to when an AI
# tier (transcription/correlation) transitions into or out of degraded
# state. Empty disables the POST entirely — not every self-hosted
# deployment will set this up, and skipping it must be silent.
ai_alert_webhook_url: str = ""
class Config:
env_file = ".env"
settings = Settings()