""" One place every Gemini call goes through: JSON-mode generation, an explicit thinking level, and token accounting. Why it exists: a day of replay runs (server-26#170) cost ~$5 of Gemini for ~7 two-hour windows — roughly $0.70 per 290 calls, which projects to several dollars a day per live deployment for correlation alone — and nothing in DRB could say where it went (server-26#45). Gemini 3.x models "think" by default and bill that as output; the old google-generativeai SDK these calls used cannot even set a thinking level. A link/new/orphan choice or a transcript cleanup does not need extended reasoning. Every call logs its token counts, and inside a replay run they are also added to the run's own usage sink (see app/internal/replay.py), so a run reports what it actually spent instead of an estimate. """ import json import threading from contextvars import ContextVar from typing import Optional from app.config import settings from app.internal.logger import logger _client = None _client_lock = threading.Lock() # Models that rejected a thinking level: retried without one from then on. _no_thinking_level: set[str] = set() _usage_sink: ContextVar[Optional[dict]] = ContextVar("drb_gemini_usage", default=None) def collect_usage(sink: Optional[dict]): """Route token counts for the current context into `sink` (a replay run). Returns a reset token.""" return _usage_sink.set(sink) def reset_usage(token) -> None: _usage_sink.reset(token) def _get_client(): global _client with _client_lock: if _client is None: from google import genai # lazy — only when a Gemini call is made _client = genai.Client(api_key=settings.gemini_api_key) return _client def _config(thinking_level: Optional[str]): from google.genai import types kwargs = {"response_mime_type": "application/json"} if thinking_level: kwargs["thinking_config"] = types.ThinkingConfig(thinking_level=thinking_level) return types.GenerateContentConfig(**kwargs) def _record(purpose: str, model: str, usage) -> None: prompt = getattr(usage, "prompt_token_count", None) or 0 output = getattr(usage, "candidates_token_count", None) or 0 thoughts = getattr(usage, "thoughts_token_count", None) or 0 logger.info(f"gemini usage {purpose} {model}: in={prompt} out={output} thinking={thoughts}") sink = _usage_sink.get() if sink is not None: row = sink.setdefault(f"{purpose}:{model}", {"calls": 0, "in": 0, "out": 0, "thinking": 0}) row["calls"] += 1 row["in"] += prompt row["out"] += output row["thinking"] += thoughts def generate_json(model: str, prompt: str, *, purpose: str, thinking_level: Optional[str] = "minimal") -> dict: """ Synchronous (run it via asyncio.to_thread). Returns the parsed JSON body. Raises on API failure, exactly like the old per-module helpers, so callers' ai_health classification (billing / dead model / transient) is unchanged. """ client = _get_client() level = None if model in _no_thinking_level else thinking_level try: resp = client.models.generate_content(model=model, contents=prompt, config=_config(level)) except Exception as e: # A model that doesn't accept this thinking level answers 400 for # every call; drop the setting for that model rather than lose the tier. if level and "thinking" in str(e).lower(): logger.warning(f"gemini: {model} rejected thinking_level={level!r} ({e}); retrying without it") _no_thinking_level.add(model) resp = client.models.generate_content(model=model, contents=prompt, config=_config(None)) else: raise _record(purpose, model, getattr(resp, "usage_metadata", None)) return json.loads(resp.text)