From 1e31b55338ffddb8a109138a3274599330cb7367 Mon Sep 17 00:00:00 2001 From: Amir Fathi Date: Sat, 25 Jul 2026 14:33:53 +0000 Subject: [PATCH] fix(agent): clamp context_percent to 100 when the window is a guess resolve_context_window() returns None when a model is not in litellm's or OpenRouter's context-window table, and the caller falls back to context_window_tokens (65536 by default). context_percent was then computed against that guessed denominator even when real usage legitimately exceeded it while staying well under the model's true, larger window, so the TUI context gauge could read past 100% while requests kept succeeding. Clamp the percent to 100 so an unresolved window never renders as over budget. context_max itself is unchanged (still surfaced as-is for the used/max label); only the derived percent is bounded. Adds a regression test that drives the real per-turn usage path with a guessed window smaller than actual usage and asserts context_percent stays at 100 instead of the raw (unclamped) ratio. Fixes #124 Co-authored-by: Claude (claude-sonnet-5) --- raven/agent/loop/main.py | 7 ++++++- tests/test_agent_loop_usage_sink.py | 28 ++++++++++++++++++++++++++++ 2 files changed, 34 insertions(+), 1 deletion(-) diff --git a/raven/agent/loop/main.py b/raven/agent/loop/main.py index f6d7661..375e2ba 100644 --- a/raven/agent/loop/main.py +++ b/raven/agent/loop/main.py @@ -2012,7 +2012,12 @@ async def _run_agent_loop( usage_sink["cost_usd"] = usage_snapshot.estimated_cost_usd usage_sink["context_max"] = context_max usage_sink["context_used"] = context_used - usage_sink["context_percent"] = round(100 * context_used / context_max) if context_max else 0 + # context_max is a real provider window when resolve_context_window() + # succeeds, but otherwise a guessed default (self.context_window_tokens) + # that can be smaller than the model's true window. Clamp so the gauge + # never reports past full: an unresolved model must not render as + # "over budget" while the request is still well within its real limit. + usage_sink["context_percent"] = min(100, round(100 * context_used / context_max)) if context_max else 0 # Context-window overflow recovery: the structured classifier flags # should_compress (a smaller window won't help, but eliding the bulk diff --git a/tests/test_agent_loop_usage_sink.py b/tests/test_agent_loop_usage_sink.py index 7963e59..d2723da 100644 --- a/tests/test_agent_loop_usage_sink.py +++ b/tests/test_agent_loop_usage_sink.py @@ -336,3 +336,31 @@ def test_refresh_context_window_on_an_openrouter_model_never_touches_the_network assert agent.context_window_tokens == rates.DEFAULT_CONTEXT_WINDOW_TOKENS assert counter["calls"] == 0 + + +@pytest.mark.asyncio +async def test_usage_sink_context_percent_clamps_when_window_is_a_guess(workspace): + """Usage past the configured fallback window must not render past 100%. + + ``stub`` never resolves via resolve_context_window(), so context_max here is + the configured (guessed) default. Real usage can legitimately exceed that + guess while staying under the model's true, unresolved window, so the + reported percent must clamp rather than imply an impossible over-budget bar. + """ + provider = UsageProvider("stub", prompt_tokens=6000, completion_tokens=2000) + agent = _make_agent(workspace, provider, model="stub", window=1000) + sink: dict = {} + + await agent._process_message( + TurnRequest( + origin=Origin.USER, + source=Source(channel="test", chat_id="c1", sender_id="user", chat_type=ChatType.DM), + text="hi", + ), + session_key="s1", + usage_sink=sink, + ) + + assert sink["context_max"] == 1000 + assert sink["context_used"] == 8000 + assert sink["context_percent"] == 100