A day of replay runs (server-26#170) spent ~$5 of Gemini on ~7 two-hour windows (~$0.70 per 290 calls) — several dollars a day per live deployment for correlation alone — and nothing could say where it went (#45). Gemini 3.x thinks by default and bills it as output; the deprecated google-generativeai SDK these calls used cannot set a thinking level. - app/internal/gemini.py: every Gemini call (correlation + transcript correction) goes through google-genai with JSON mode, an explicit thinking level, and logs in/out/thinking tokens. A model that rejects the level is retried without it once and remembered, so the tier is never lost to a config param. API failures still raise for ai_health. - correlator: thinking_level "minimal" (a link/new/orphan choice). transcript correction: "low" until a replay shows minimal is safe. - replay: runs record real Gemini token usage (metrics.gemini_usage), shown in the Replay tab. - requirements: google-genai. c2-core: 474 pass. Frontend typecheck not run (no Node on this box). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
95 lines
3.8 KiB
Python
95 lines
3.8 KiB
Python
"""
|
|
One place every Gemini call goes through: JSON-mode generation, an explicit
|
|
thinking level, and token accounting.
|
|
|
|
Why it exists: a day of replay runs (server-26#170) cost ~$5 of Gemini for
|
|
~7 two-hour windows — roughly $0.70 per 290 calls, which projects to several
|
|
dollars a day per live deployment for correlation alone — and nothing in DRB
|
|
could say where it went (server-26#45). Gemini 3.x models "think" by default
|
|
and bill that as output; the old google-generativeai SDK these calls used
|
|
cannot even set a thinking level. A link/new/orphan choice or a transcript
|
|
cleanup does not need extended reasoning.
|
|
|
|
Every call logs its token counts, and inside a replay run they are also added
|
|
to the run's own usage sink (see app/internal/replay.py), so a run reports
|
|
what it actually spent instead of an estimate.
|
|
"""
|
|
import json
|
|
import threading
|
|
from contextvars import ContextVar
|
|
from typing import Optional
|
|
|
|
from app.config import settings
|
|
from app.internal.logger import logger
|
|
|
|
_client = None
|
|
_client_lock = threading.Lock()
|
|
# Models that rejected a thinking level: retried without one from then on.
|
|
_no_thinking_level: set[str] = set()
|
|
|
|
_usage_sink: ContextVar[Optional[dict]] = ContextVar("drb_gemini_usage", default=None)
|
|
|
|
|
|
def collect_usage(sink: Optional[dict]):
|
|
"""Route token counts for the current context into `sink` (a replay run). Returns a reset token."""
|
|
return _usage_sink.set(sink)
|
|
|
|
|
|
def reset_usage(token) -> None:
|
|
_usage_sink.reset(token)
|
|
|
|
|
|
def _get_client():
|
|
global _client
|
|
with _client_lock:
|
|
if _client is None:
|
|
from google import genai # lazy — only when a Gemini call is made
|
|
_client = genai.Client(api_key=settings.gemini_api_key)
|
|
return _client
|
|
|
|
|
|
def _config(thinking_level: Optional[str]):
|
|
from google.genai import types
|
|
kwargs = {"response_mime_type": "application/json"}
|
|
if thinking_level:
|
|
kwargs["thinking_config"] = types.ThinkingConfig(thinking_level=thinking_level)
|
|
return types.GenerateContentConfig(**kwargs)
|
|
|
|
|
|
def _record(purpose: str, model: str, usage) -> None:
|
|
prompt = getattr(usage, "prompt_token_count", None) or 0
|
|
output = getattr(usage, "candidates_token_count", None) or 0
|
|
thoughts = getattr(usage, "thoughts_token_count", None) or 0
|
|
logger.info(f"gemini usage {purpose} {model}: in={prompt} out={output} thinking={thoughts}")
|
|
sink = _usage_sink.get()
|
|
if sink is not None:
|
|
row = sink.setdefault(f"{purpose}:{model}", {"calls": 0, "in": 0, "out": 0, "thinking": 0})
|
|
row["calls"] += 1
|
|
row["in"] += prompt
|
|
row["out"] += output
|
|
row["thinking"] += thoughts
|
|
|
|
|
|
def generate_json(model: str, prompt: str, *, purpose: str,
|
|
thinking_level: Optional[str] = "minimal") -> dict:
|
|
"""
|
|
Synchronous (run it via asyncio.to_thread). Returns the parsed JSON body.
|
|
Raises on API failure, exactly like the old per-module helpers, so callers'
|
|
ai_health classification (billing / dead model / transient) is unchanged.
|
|
"""
|
|
client = _get_client()
|
|
level = None if model in _no_thinking_level else thinking_level
|
|
try:
|
|
resp = client.models.generate_content(model=model, contents=prompt, config=_config(level))
|
|
except Exception as e:
|
|
# A model that doesn't accept this thinking level answers 400 for
|
|
# every call; drop the setting for that model rather than lose the tier.
|
|
if level and "thinking" in str(e).lower():
|
|
logger.warning(f"gemini: {model} rejected thinking_level={level!r} ({e}); retrying without it")
|
|
_no_thinking_level.add(model)
|
|
resp = client.models.generate_content(model=model, contents=prompt, config=_config(None))
|
|
else:
|
|
raise
|
|
_record(purpose, model, getattr(resp, "usage_metadata", None))
|
|
return json.loads(resp.text)
|