From b1db40cbb3252cd75d67238472185afe2e1b3ed6 Mon Sep 17 00:00:00 2001 From: uipreliga Date: Sat, 26 Sep 2026 13:01:41 -0700 Subject: [PATCH] fix(antigravity): read token usage from the conversation, not the Step google-antigravity 0.1.18 (bumped in 3aa2db30) deprecated Step.usage_metadata: usage is now billed per model invocation on the conversation (conversation.total_usage). The adapter still read the per-step field, so every Antigravity turn since that bump booked no tokens and no cost, and api_calls stayed at 1 because only a usage-bearing step closes a call, so max_turns never tripped. Nothing failed loudly; make verify stayed green. The turn state now takes a cumulative_usage source, books each rise since the last reading as one generation, and books any remainder at finalize, so the per-message buckets still sum to the turn total. The test fakes bill through a meter the way the SDK connection does, and no longer put usage on the Step, so the suite exercises the new source. Verified live on gemini-3.8-flash: 30,774 input / 803 output tokens, 3 generations, 3 counted calls, $0.052 at list price (previously None). Co-Authored-By: Claude Opus 5.5 (1M context) --- .claude/harness-candidates.md | 10 +++ .claude/notes/timing.md | 4 +- docs/agents/HARNESS_PARITY.md | 9 +-- src/coder_eval/agents/antigravity_agent.py | 53 +++++++++++++--- .../golden_streams/antigravity_fixtures.py | 33 +++++++++- tests/test_antigravity_agent.py | 61 ++++++++++++++++++- tests/test_early_stop.py | 2 + tests/test_timing_identity_contract.py | 10 ++- 8 files changed, 158 insertions(+), 24 deletions(-) diff --git a/.claude/harness-candidates.md b/.claude/harness-candidates.md index 4880e2642..b6fb6fb3a 100644 --- a/.claude/harness-candidates.md +++ b/.claude/harness-candidates.md @@ -1022,3 +1022,13 @@ re-derive from scratch. have, and it is a recurring class ONLY within one file so far — worth promoting to a rule if a second subprocess-driven agent repeats it. Caught in: final cross-phase review, delegate agent port. + +## From 2026-09-26 Antigravity SDK 0.1.18 usage regression + +- [ ] Runtime guard: fail an Antigravity turn that ran MODEL steps but booked zero + tokens, as OpenCode's `require_token_telemetry` does. SDK 0.1.18 deprecated + `Step.usage_metadata` (usage moved to `conversation.total_usage`), so every + Antigravity turn from `3aa2db30` on reported no tokens and no cost, and `api_calls` + stayed at 1 so `max_turns` never tripped — all silently, with `make verify` green. + A static rule cannot see an SDK field go dead; a live-telemetry smoke (one real turn, + assert `token_usage` is non-empty) in the harness-bump checklist would have. diff --git a/.claude/notes/timing.md b/.claude/notes/timing.md index 879fb0a48..13a9b8b99 100644 --- a/.claude/notes/timing.md +++ b/.claude/notes/timing.md @@ -401,8 +401,8 @@ and the tail are already computed. A reducer's only remaining timing decision is where its window opens, which is the one genuinely harness-shaped part: two interleave a tool into a single window outright -(Antigravity, whose Step for the tool arrives and only a later `usage_metadata` Step cuts -the message, and Codex, whose `_flush_message` window extends to the last item's +(Antigravity, whose Step for the tool arrives and only a later rise in the conversation's +billed usage cuts the message, and Codex, whose `_flush_message` window extends to the last item's `completed_at_ms`) while the other three tile the turn contiguously, so a call open at a boundary runs inside two windows. Central subtraction handles both without either reducer knowing which it is. diff --git a/docs/agents/HARNESS_PARITY.md b/docs/agents/HARNESS_PARITY.md index f473be81f..86fb07bb4 100644 --- a/docs/agents/HARNESS_PARITY.md +++ b/docs/agents/HARNESS_PARITY.md @@ -67,8 +67,8 @@ All five harnesses can have tool execution inside a generation window, and it is subtracted out of every one of them — **once, centrally**, by `timing.py::subtract_tool_time`. No reducer does it itself; each publishes the raw window (see the two sections below). Every harness has the -problem: Antigravity reports a `Step` for the tool and only a later -`usage_metadata` `Step` cuts the message; Codex's message window is seeded from +problem: Antigravity reports a `Step` for the tool and only a later rise in +the conversation's billed usage cuts the message; Codex's message window is seeded from the first item's start and extended to the last item's completion; OpenCode, Pi and claude-code tile, each window opening where the previous one closed and running to the next, with every tool call in between running inside. In each the @@ -559,8 +559,9 @@ Each harness finds the call boundary in its own stream (the table above): first item would let one tool of call N+1 act. A call that ran tools therefore opens the next call at its `tokenUsage`, since the results always go back to the model, and the cap fires before call N+1 can run anything. -- **Antigravity** attaches `usage_metadata` to one step per call, and the next call - opens with a MODEL step at a new `step_index`. +- **Antigravity** bills each call on the conversation's cumulative usage, not on a + step, so a rise in that total closes the call, and the next call opens with a MODEL + step at a new `step_index`. - **OpenCode** and **Pi** stream one `step_start` or `turn_start` per call. - **Delegate**'s SDK has no round-trip marker, and a tool-only reply streams only its tool call, with no text before it. So the next call opens when every tool the diff --git a/src/coder_eval/agents/antigravity_agent.py b/src/coder_eval/agents/antigravity_agent.py index b77de618d..5c82e488e 100644 --- a/src/coder_eval/agents/antigravity_agent.py +++ b/src/coder_eval/agents/antigravity_agent.py @@ -25,7 +25,7 @@ from contextlib import AsyncExitStack from datetime import datetime from pathlib import Path -from typing import Any, ClassVar +from typing import Any, ClassVar, NamedTuple from coder_eval.agent import Agent, AgentState from coder_eval.agents._logging import PrefixedAdapter @@ -152,6 +152,22 @@ def _enum_value(x: Any) -> Any: return getattr(x, "value", x) +class _UsageCounts(NamedTuple): + """The four Gemini token counts, named as ``google.antigravity.types.UsageMetadata`` names them.""" + + prompt_token_count: int = 0 + cached_content_token_count: int = 0 + candidates_token_count: int = 0 + thoughts_token_count: int = 0 + + @classmethod + def of(cls, usage: Any) -> "_UsageCounts": + return cls(*((getattr(usage, f, 0) or 0) for f in cls._fields)) + + def minus(self, other: "_UsageCounts") -> "_UsageCounts": + return _UsageCounts(*(a - b for a, b in zip(self, other, strict=True))) + + def _to_token_usage(usage: Any, model: str | None) -> TokenUsage: """Map a ``google.antigravity.types.UsageMetadata`` to coder_eval ``TokenUsage``. @@ -480,6 +496,7 @@ async def communicate( model=model, turn_start_time=turn_start_time, clock=clock, + cumulative_usage=lambda: self._sdk_agent.conversation.total_usage, max_turns=max_turns, ) @@ -677,10 +694,13 @@ class _AntigravityTurnState: agent's shared crash kernel builds ``pending_turn`` from ``collector``). Step-stream shape this consumes (observed): each ``step_index`` is yielded - repeatedly through ACTIVE -> DONE transitions; ``usage_metadata`` lands once - per generation on a DONE/terminal step (summing them == the turn total); a - tool call carries a stable ``id`` and its result is folded into expanded - ``args`` at DONE. + repeatedly through ACTIVE -> DONE transitions; a tool call carries a stable + ``id`` and its result is folded into expanded ``args`` at DONE. + + Usage is NOT on the Step: the SDK accumulates it per model invocation on the + conversation, which ``cumulative_usage`` reads. Each rise since the last + reading is booked as one generation, and ``finalize`` books the remainder, so + the per-message buckets always sum to the turn total. """ def __init__( @@ -696,9 +716,12 @@ def __init__( model: str, turn_start_time: float, clock: TurnClock, + cumulative_usage: Callable[[], Any], max_turns: int | None = None, ) -> None: self._agent = agent + self._cumulative_usage = cumulative_usage + self._usage_booked = _UsageCounts.of(cumulative_usage()) self.emit = emit self.task_id = task_id self.turn_id = turn_id @@ -815,13 +838,22 @@ def process_step(self, step: Any) -> None: self._blocks.append(ContentBlock(block_type="text", sequence=0, text=step.content)) # Per-generation usage: fold into the turn total and cut an AssistantMessage. - if step.usage_metadata is not None: + if self._book_new_usage(): if not self._in_api_call: self.api_calls += 1 self._in_api_call = False - gen = _to_token_usage(step.usage_metadata, self.model) - self.total_usage = self.total_usage + gen - self._flush_generation(gen, getattr(step.usage_metadata, "thoughts_token_count", 0) or 0) + + def _book_new_usage(self) -> bool: + """Book the usage accrued since the last reading as one generation; False if none.""" + now = _UsageCounts.of(self._cumulative_usage()) + accrued = now.minus(self._usage_booked) + if not any(accrued): + return False + self._usage_booked = now + gen = _to_token_usage(accrued, self.model) + self.total_usage = self.total_usage + gen + self._flush_generation(gen, accrued.thoughts_token_count) + return True def _handle_tool_call(self, call: Any, step: Any, done: bool, sstatus: Any, call_index: int) -> None: raw_name = _enum_value(call.name) @@ -995,7 +1027,8 @@ def finalize(self, status: AgentEndStatus, *, crashed: bool = False, crash_reaso ) ) - # Flush any trailing blocks not yet attached to a generation (no usage). + # Usage that landed after the last Step, then any blocks it did not carry. + self._book_new_usage() if self._blocks: self._flush_generation(TokenUsage(), 0) diff --git a/tests/_fixtures/golden_streams/antigravity_fixtures.py b/tests/_fixtures/golden_streams/antigravity_fixtures.py index 17f4ccf86..4ca1587dd 100644 --- a/tests/_fixtures/golden_streams/antigravity_fixtures.py +++ b/tests/_fixtures/golden_streams/antigravity_fixtures.py @@ -66,7 +66,9 @@ def _step( content_delta=content_delta, thinking=thinking, thinking_delta=thinking_delta, - usage_metadata=usage, + # NOT `usage_metadata`: SDK 0.1.18 no longer puts usage on the Step. The fake + # conversation's meter bills it as the step is yielded, as the connection does. + billed=usage, is_complete_response=complete, error=error, step_index=step_index, @@ -74,6 +76,28 @@ def _step( ) +class _UsageMeter: + """Stands in for the SDK connection's ``cumulative_usage``: rises as model calls are billed.""" + + def __init__(self) -> None: + self.total = _usage(0, 0, 0, 0) + + def feed(self, step): + billed = getattr(step, "billed", None) + if billed is not None: + t = self.total + self.total = _usage( + t.prompt_token_count + billed.prompt_token_count, + t.cached_content_token_count + billed.cached_content_token_count, + t.candidates_token_count + billed.candidates_token_count, + t.thoughts_token_count + billed.thoughts_token_count, + ) + return step + + def read(self): + return self.total + + class _FakeConversation: """Scriptable fake SDK conversation. @@ -94,6 +118,11 @@ def __init__(self, steps): self.last_response = "" self.receive_steps_call_count = 0 self.cancel_call_count = 0 + self.meter = _UsageMeter() + + @property + def total_usage(self): + return self.meter.read() async def send(self, prompt, **kwargs): return None @@ -103,7 +132,7 @@ async def receive_steps(self): batch = self._batches[self._batch_index] if self._batch_index < len(self._batches) else [] self._batch_index += 1 for s in batch: - yield s + yield self.meter.feed(s) async def cancel(self): self.cancel_call_count += 1 diff --git a/tests/test_antigravity_agent.py b/tests/test_antigravity_agent.py index 35b747340..5c0e963c6 100644 --- a/tests/test_antigravity_agent.py +++ b/tests/test_antigravity_agent.py @@ -32,10 +32,12 @@ from tests._fixtures.golden_streams._scrub import assert_reconciliation from tests._fixtures.golden_streams.antigravity_fixtures import ( _agent_with_steps, + _FakeConversation, _no_sleep, _step, _tc, _usage, + _UsageMeter, ) @@ -322,6 +324,43 @@ async def test_communicate_maps_steps_to_turn_record(): assert agent.pending_turn is None # success path leaves no partial +async def test_usage_comes_from_the_conversation_not_the_step(): + """SDK 0.1.18 stopped filling ``Step.usage_metadata``; reading it booked every turn at zero tokens.""" + step = _step("TEXT_RESPONSE", "DONE", content="done", complete=True) + step.usage_metadata = _usage(999, 0, 99, 0) + agent = _agent_with_steps([step]) + + tr = await agent.communicate("go") + + assert tr.token_usage is None + + +async def test_usage_billed_after_the_last_step_is_still_booked(): + """The conversation may bill a call after its last Step; finalize books the remainder.""" + + class _BillsAfterTheLastStep(_FakeConversation): + async def receive_steps(self): + async for step in super().receive_steps(): + yield step + self.meter.feed(SimpleNamespace(billed=_usage(40, 0, 7, 0))) + + steps = [ + _step("THINKING", "DONE", thinking="plan", usage=_usage(100, 0, 5, 5)), + _step("TEXT_RESPONSE", "DONE", content="done", content_delta="done", complete=True, step_index=1), + ] + agent = _agent_with_steps([]) + agent._sdk_agent = SimpleNamespace(conversation=_BillsAfterTheLastStep(steps), is_started=True) + + tr = await agent.communicate("go") + + assert tr.token_usage.uncached_input_tokens == 100 + 40 + assert tr.token_usage.output_tokens == (5 + 5) + 7 + assert tr.num_turns == 2 + bucketed = [m for m in tr.messages if hasattr(m, "cache_creation_tokens")] + assert sum(m.input_tokens for m in bucketed) == tr.token_usage.uncached_input_tokens + assert sum(m.output_tokens for m in bucketed) == tr.token_usage.output_tokens + + async def test_communicate_normalizes_arg_keys_and_strips_done_only_results(): """LS directory_path -> path, and tool-specific result fields that first appear at DONE (LS ``results``, WebSearch ``summary``) are stripped from @@ -380,6 +419,7 @@ async def test_communicate_crash_sets_pending_partial_turn(): class _Boom: last_response = "" + total_usage = _usage(0, 0, 0, 0) async def send(self, prompt, **kwargs): return None @@ -419,6 +459,7 @@ async def test_communicate_timeout_sets_pending_partial_turn(monkeypatch): class _Cancelled: last_response = "" + total_usage = _usage(0, 0, 0, 0) async def send(self, prompt, **kwargs): return None @@ -982,6 +1023,11 @@ def __init__(self, batches): self._batch_index = 0 self._is_receiving = False # lives on the "connection" layer, like the real SDK self.receive_steps_call_count = 0 + self.meter = _UsageMeter() + + @property + def total_usage(self): + return self.meter.read() async def send(self, prompt, **kwargs): return None @@ -994,7 +1040,7 @@ async def _connection_receive_steps(self): batch = self._batches[self._batch_index] if self._batch_index < len(self._batches) else [] self._batch_index += 1 for s in batch: - yield s + yield self.meter.feed(s) finally: self._is_receiving = False @@ -1070,6 +1116,11 @@ class _FiresWatchdogThenCancelsOnSecondDrain: def __init__(self) -> None: self.call_count = 0 + self.meter = _UsageMeter() + + @property + def total_usage(self): + return self.meter.read() async def send(self, prompt, **kwargs): return None @@ -1083,7 +1134,9 @@ async def receive_steps(self): target="TARGET_ENVIRONMENT", tool_calls=[_tc("run_command", "bg1", {"command_line": "sleep 999"})], ) - yield _step("TEXT_RESPONSE", "DONE", content="started", complete=True, usage=_usage(10, 0, 1, 0)) + yield self.meter.feed( + _step("TEXT_RESPONSE", "DONE", content="started", complete=True, usage=_usage(10, 0, 1, 0)) + ) else: assert _WatchdogFiresLater.captured_on_timeout is not None _WatchdogFiresLater.captured_on_timeout() @@ -2065,6 +2118,7 @@ def _state(self, clock): agent = AntigravityAgent(parse_agent_config(type="antigravity", model="gemini-3.5-flash")) collector = EventCollector() + self.meter = _UsageMeter() return _AntigravityTurnState( agent=agent, emit=CompositeStreamCallback([collector]), @@ -2076,6 +2130,7 @@ def _state(self, clock): model="gemini-3.5-flash", turn_start_time=0.0, clock=clock, + cumulative_usage=self.meter.read, ) def test_the_first_step_moves_the_mark_off_the_turn_entry_stamp(self): @@ -2127,7 +2182,7 @@ def test_a_flush_still_advances_the_mark_and_opens_at_the_reseeded_one(self): clock.at_ms = 900 state.process_step(_step("THINKING", "ACTIVE", thinking="plan")) clock.at_ms = 2000 - state.process_step(_step("THINKING", "DONE", thinking="plan", usage=_usage(100, 0, 5, 5))) + state.process_step(self.meter.feed(_step("THINKING", "DONE", thinking="plan", usage=_usage(100, 0, 5, 5)))) message = _assistant(state)[0] assert message.started_at == self.BASE + timedelta(milliseconds=900), "opens at the RE-SEEDED mark" diff --git a/tests/test_early_stop.py b/tests/test_early_stop.py index ad75bc27d..431b68f9e 100644 --- a/tests/test_early_stop.py +++ b/tests/test_early_stop.py @@ -2990,6 +2990,8 @@ def __init__(self, steps: list[Any], *, cancel_raises: bool = False) -> None: self.yielded = 0 self.cancels = 0 self.last_response = "" + # These streams bill nothing; the seam under test is the stop, not the usage. + self.total_usage = SimpleNamespace() async def send(self, prompt: Any, **_kwargs: Any) -> None: return None diff --git a/tests/test_timing_identity_contract.py b/tests/test_timing_identity_contract.py index f2548d72b..6ee23b9e5 100644 --- a/tests/test_timing_identity_contract.py +++ b/tests/test_timing_identity_contract.py @@ -322,11 +322,12 @@ def _antigravity_turn() -> Turn: measurable head at all. """ from coder_eval.agents.antigravity_agent import AntigravityAgent, _AntigravityTurnState - from tests._fixtures.golden_streams.antigravity_fixtures import _step, _tc, _usage + from tests._fixtures.golden_streams.antigravity_fixtures import _step, _tc, _usage, _UsageMeter agent = AntigravityAgent(parse_agent_config(type=AgentKind.ANTIGRAVITY, model="gemini-3.5-flash")) collector = EventCollector() clock = _InjectedClock(at_ms=500) # dispatch before the first Step: head + meter = _UsageMeter() state = _AntigravityTurnState( agent=agent, emit=CompositeStreamCallback([collector]), @@ -338,6 +339,7 @@ def _antigravity_turn() -> Turn: model="gemini-3.5-flash", turn_start_time=0.0, clock=clock, + cumulative_usage=meter.read, ) clock.at_ms = 700 @@ -359,9 +361,11 @@ def _antigravity_turn() -> Turn: ) ) clock.at_ms = 2000 - state.process_step(_step("THINKING", "DONE", thinking="plan", usage=_usage(100, 0, 5, 5))) + state.process_step(meter.feed(_step("THINKING", "DONE", thinking="plan", usage=_usage(100, 0, 5, 5)))) clock.at_ms = 3000 - state.process_step(_step("TEXT_RESPONSE", "DONE", content="done", complete=True, usage=_usage(200, 0, 10, 0))) + state.process_step( + meter.feed(_step("TEXT_RESPONSE", "DONE", content="done", complete=True, usage=_usage(200, 0, 10, 0))) + ) return Turn(started_ms=0.0, ended_ms=3500.0, messages=list(state.messages), commands=list(state.commands))