gemini: minimal thinking on correlation, token accounting per call
A day of replay runs (server-26#170) spent ~$5 of Gemini on ~7 two-hour windows (~$0.70 per 290 calls) — several dollars a day per live deployment for correlation alone — and nothing could say where it went (#45). Gemini 3.x thinks by default and bills it as output; the deprecated google-generativeai SDK these calls used cannot set a thinking level. - app/internal/gemini.py: every Gemini call (correlation + transcript correction) goes through google-genai with JSON mode, an explicit thinking level, and logs in/out/thinking tokens. A model that rejects the level is retried without it once and remembered, so the tier is never lost to a config param. API failures still raise for ai_health. - correlator: thinking_level "minimal" (a link/new/orphan choice). transcript correction: "low" until a replay shows minimal is safe. - replay: runs record real Gemini token usage (metrics.gemini_usage), shown in the Replay tab. - requirements: google-genai. c2-core: 474 pass. Frontend typecheck not run (no Node on this box). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
266c958208
commit
0543526eb0
@@ -0,0 +1,70 @@
|
||||
"""
|
||||
app/internal/gemini.py — thinking level, fallback when a model rejects it,
|
||||
and token accounting into a replay's usage sink (server-26#170 cost finding).
|
||||
"""
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import patch
|
||||
|
||||
from app.internal import gemini
|
||||
|
||||
|
||||
class _FakeModels:
|
||||
def __init__(self, reject_thinking=False):
|
||||
self.reject_thinking = reject_thinking
|
||||
self.configs = []
|
||||
|
||||
def generate_content(self, model, contents, config):
|
||||
self.configs.append(config)
|
||||
if self.reject_thinking and config.get("thinking_level"):
|
||||
raise RuntimeError("400 INVALID_ARGUMENT: thinking_level is not supported for this model")
|
||||
return SimpleNamespace(
|
||||
text='{"action": "link"}',
|
||||
usage_metadata=SimpleNamespace(prompt_token_count=1200, candidates_token_count=30,
|
||||
thoughts_token_count=0),
|
||||
)
|
||||
|
||||
|
||||
def _patched(models):
|
||||
client = SimpleNamespace(models=models)
|
||||
return (patch.object(gemini, "_get_client", return_value=client),
|
||||
patch.object(gemini, "_config", lambda level: {"thinking_level": level}))
|
||||
|
||||
|
||||
def test_minimal_thinking_by_default_and_usage_lands_in_the_sink():
|
||||
models = _FakeModels()
|
||||
a, b = _patched(models)
|
||||
sink = {}
|
||||
tok = gemini.collect_usage(sink)
|
||||
try:
|
||||
with a, b:
|
||||
assert gemini.generate_json("m1", "p", purpose="correlation") == {"action": "link"}
|
||||
finally:
|
||||
gemini.reset_usage(tok)
|
||||
assert models.configs == [{"thinking_level": "minimal"}]
|
||||
assert sink == {"correlation:m1": {"calls": 1, "in": 1200, "out": 30, "thinking": 0}}
|
||||
|
||||
|
||||
def test_model_that_rejects_thinking_level_falls_back_once():
|
||||
models = _FakeModels(reject_thinking=True)
|
||||
a, b = _patched(models)
|
||||
gemini._no_thinking_level.discard("m2")
|
||||
with a, b:
|
||||
gemini.generate_json("m2", "p", purpose="correlation")
|
||||
gemini.generate_json("m2", "p", purpose="correlation")
|
||||
# first call: tried minimal, retried without; second call: straight without
|
||||
assert models.configs == [{"thinking_level": "minimal"}, {"thinking_level": None}, {"thinking_level": None}]
|
||||
gemini._no_thinking_level.discard("m2")
|
||||
|
||||
|
||||
def test_other_failures_still_raise_for_ai_health():
|
||||
class Boom(_FakeModels):
|
||||
def generate_content(self, **kw):
|
||||
raise RuntimeError("429 insufficient_quota")
|
||||
a, b = _patched(Boom())
|
||||
with a, b:
|
||||
try:
|
||||
gemini.generate_json("m3", "p", purpose="correlation")
|
||||
except RuntimeError as e:
|
||||
assert "insufficient_quota" in str(e)
|
||||
else:
|
||||
raise AssertionError("should raise")
|
||||
Reference in New Issue
Block a user