A day of replay runs (server-26#170) spent ~$5 of Gemini on ~7 two-hour windows (~$0.70 per 290 calls) — several dollars a day per live deployment for correlation alone — and nothing could say where it went (#45). Gemini 3.x thinks by default and bills it as output; the deprecated google-generativeai SDK these calls used cannot set a thinking level. - app/internal/gemini.py: every Gemini call (correlation + transcript correction) goes through google-genai with JSON mode, an explicit thinking level, and logs in/out/thinking tokens. A model that rejects the level is retried without it once and remembered, so the tier is never lost to a config param. API failures still raise for ai_health. - correlator: thinking_level "minimal" (a link/new/orphan choice). transcript correction: "low" until a replay shows minimal is safe. - replay: runs record real Gemini token usage (metrics.gemini_usage), shown in the Replay tab. - requirements: google-genai. c2-core: 474 pass. Frontend typecheck not run (no Node on this box). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
71 lines
2.6 KiB
Python
71 lines
2.6 KiB
Python
"""
|
|
app/internal/gemini.py — thinking level, fallback when a model rejects it,
|
|
and token accounting into a replay's usage sink (server-26#170 cost finding).
|
|
"""
|
|
from types import SimpleNamespace
|
|
from unittest.mock import patch
|
|
|
|
from app.internal import gemini
|
|
|
|
|
|
class _FakeModels:
|
|
def __init__(self, reject_thinking=False):
|
|
self.reject_thinking = reject_thinking
|
|
self.configs = []
|
|
|
|
def generate_content(self, model, contents, config):
|
|
self.configs.append(config)
|
|
if self.reject_thinking and config.get("thinking_level"):
|
|
raise RuntimeError("400 INVALID_ARGUMENT: thinking_level is not supported for this model")
|
|
return SimpleNamespace(
|
|
text='{"action": "link"}',
|
|
usage_metadata=SimpleNamespace(prompt_token_count=1200, candidates_token_count=30,
|
|
thoughts_token_count=0),
|
|
)
|
|
|
|
|
|
def _patched(models):
|
|
client = SimpleNamespace(models=models)
|
|
return (patch.object(gemini, "_get_client", return_value=client),
|
|
patch.object(gemini, "_config", lambda level: {"thinking_level": level}))
|
|
|
|
|
|
def test_minimal_thinking_by_default_and_usage_lands_in_the_sink():
|
|
models = _FakeModels()
|
|
a, b = _patched(models)
|
|
sink = {}
|
|
tok = gemini.collect_usage(sink)
|
|
try:
|
|
with a, b:
|
|
assert gemini.generate_json("m1", "p", purpose="correlation") == {"action": "link"}
|
|
finally:
|
|
gemini.reset_usage(tok)
|
|
assert models.configs == [{"thinking_level": "minimal"}]
|
|
assert sink == {"correlation:m1": {"calls": 1, "in": 1200, "out": 30, "thinking": 0}}
|
|
|
|
|
|
def test_model_that_rejects_thinking_level_falls_back_once():
|
|
models = _FakeModels(reject_thinking=True)
|
|
a, b = _patched(models)
|
|
gemini._no_thinking_level.discard("m2")
|
|
with a, b:
|
|
gemini.generate_json("m2", "p", purpose="correlation")
|
|
gemini.generate_json("m2", "p", purpose="correlation")
|
|
# first call: tried minimal, retried without; second call: straight without
|
|
assert models.configs == [{"thinking_level": "minimal"}, {"thinking_level": None}, {"thinking_level": None}]
|
|
gemini._no_thinking_level.discard("m2")
|
|
|
|
|
|
def test_other_failures_still_raise_for_ai_health():
|
|
class Boom(_FakeModels):
|
|
def generate_content(self, **kw):
|
|
raise RuntimeError("429 insufficient_quota")
|
|
a, b = _patched(Boom())
|
|
with a, b:
|
|
try:
|
|
gemini.generate_json("m3", "p", purpose="correlation")
|
|
except RuntimeError as e:
|
|
assert "insufficient_quota" in str(e)
|
|
else:
|
|
raise AssertionError("should raise")
|