diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 4fe3c9f752..e8f466bc09 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -610,15 +610,23 @@ def _image_error_max_dimension(error: Exception) -> Optional[int]: def _pressure_with_real_floor(compressor: Any, rough_tokens: int) -> int: - """Floor the rough pre-API pressure estimate at the last REAL prompt size. + """Floor the ROUGH pre-API pressure estimate at the last REAL prompt size. - The chars/4 rough estimate under-counts Cyrillic (and other non-ASCII - scripts) by up to ~2x, so a session can sit at the provider's real context - ceiling while the rough figure stays under the compaction threshold — on - silent-clip providers (ollama /v1) that is a truncation death spiral the - reactive overflow handler never sees (observed live: real prompts - 64,842→64,995 against a 55,705 threshold). The provider's own last - reported prompt_tokens is authoritative; never let the pressure figure + Applied only on the fallback path -- when ``anchored_context_tokens`` has + no valid anchor (first request, transcript rewritten under the anchor, + provider never reported usage). A valid anchor is provider-exact and is + used as-is; in particular on MoA turns the anchor deliberately uses the + pre-fold aggregator usage while ``last_real_prompt_tokens`` holds the + folded figure, so flooring an anchored value would re-add fan-out tokens + the anchor exists to exclude. + + On the rough path, non-ASCII text (Cyrillic, Greek, Polish, ...) + under-counts by up to ~2x, so a session can sit at the provider's real + context ceiling while the rough figure stays under the compaction + threshold -- on silent-clip providers (ollama /v1) that is a truncation + death spiral the reactive overflow handler never sees (observed live: + real prompts 64,842->64,995 against a 55,705 threshold). The provider's + last reported prompt_tokens is authoritative; never let the rough figure fall below it. Skipped for exactly one turn after a compaction, when last_real_prompt_tokens still holds the stale pre-compression value (#36718's awaiting_real_usage_after_compression window). @@ -2904,9 +2912,10 @@ def run_conversation( ) if _anchored_pressure is not None: request_pressure_tokens = _anchored_pressure - request_pressure_tokens = _pressure_with_real_floor( - agent.context_compressor, request_pressure_tokens - ) + else: + request_pressure_tokens = _pressure_with_real_floor( + agent.context_compressor, request_pressure_tokens + ) total_chars = approx_tokens * 4 # Stash this request's rough estimate so update_from_response() can # pair it with the provider's real prompt count — the (rough, real) diff --git a/agent/model_metadata.py b/agent/model_metadata.py index d93055252a..052e1a1e4e 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -3662,10 +3662,19 @@ def estimate_tokens_rough(text: str) -> int: # token) where chars/4 under-counted them ~2x and let sessions ride # the provider's context ceiling below the compaction threshold. # ASCII spans inside mixed text still count at 1 byte each. - return (len(text.encode("utf-8")) + 3) // 4 + # + # Calibrated against cl100k/o200k/Qwen2.5 (estimate / mean real): + # Russian 0.67->1.24, Ukrainian 0.55->1.03, Arabic 0.53->0.96, + # Hindi 0.34->0.90, Greek 0.37->0.68, Polish 0.63->0.69; accented + # Latin barely moves (French 1.02->1.03, German 0.99->1.02, + # Spanish 1.04->1.07) because only the accented chars widen. + # Pure-ASCII prose already over-counts at ~1.4 on the same rule. + # errors="replace": lone surrogates (routine in tool output; see + # message_sanitization) must not turn an estimate into a raise. + return (len(text.encode("utf-8", "replace")) + 3) // 4 # Mixed CJK + other: dense chars stay ~1 token each; the sparse # remainder is byte-counted for the same corrective. - return dense + ((len(stripped.encode("utf-8")) + 3) // 4) + return dense + ((len(stripped.encode("utf-8", "replace")) + 3) // 4) def estimate_messages_tokens_rough( diff --git a/tests/agent/test_cjk_token_estimation.py b/tests/agent/test_cjk_token_estimation.py index c584597ad6..3c39c54cd0 100644 --- a/tests/agent/test_cjk_token_estimation.py +++ b/tests/agent/test_cjk_token_estimation.py @@ -88,3 +88,32 @@ def test_cyrillic_counts_by_utf8_bytes(): assert estimate_tokens_rough("русский текст") == 7 # Pure ASCII unchanged. assert estimate_tokens_rough("a" * 400) == 100 + + +def test_accented_latin_is_not_inflated_by_byte_counting(): + # Byte-counting must not punish Western-European text: only the accented + # chars are 2 bytes, so the estimate moves by a few percent, not 2x. + from agent.model_metadata import estimate_tokens_rough + fr = "La compression du contexte permet aux longues sessions de rester dans la fenêtre du fournisseur sans perdre le fil de la tâche." + ascii_rule = (len(fr) + 3) // 4 + est = estimate_tokens_rough(fr) + assert ascii_rule <= est <= int(ascii_rule * 1.10), (ascii_rule, est) + + +def test_mixed_cyrillic_and_ascii_code_counts_ascii_at_one_byte(): + from agent.model_metadata import estimate_tokens_rough + code = "def compress(ctx):\n # Сжимаем контекст\n return summarize(ctx)\n" + ascii_part = "def compress(ctx):\n # \n return summarize(ctx)\n" + cyr = "Сжимаем контекст" + expected = (len(ascii_part.encode()) + len(cyr.encode()) + 3) // 4 + assert estimate_tokens_rough(code) == expected + # and strictly more than the old chars/4 rule for the same text + assert estimate_tokens_rough(code) > (len(code) + 3) // 4 + + +def test_lone_surrogates_do_not_raise(): + # main's estimator was total (len/regex never raise); byte-counting must + # stay total too — tool output routinely carries unpaired surrogates. + from agent.model_metadata import estimate_tokens_rough + assert estimate_tokens_rough("abc\ud800def") >= 2 + assert estimate_tokens_rough("漢字\udfff") >= 2 diff --git a/tests/run_agent/test_infinite_compaction_loop.py b/tests/run_agent/test_infinite_compaction_loop.py index 6082cddff4..9eb0a84de5 100644 --- a/tests/run_agent/test_infinite_compaction_loop.py +++ b/tests/run_agent/test_infinite_compaction_loop.py @@ -292,3 +292,22 @@ class TestPressureRealFloor: from agent.conversation_loop import _pressure_with_real_floor assert _pressure_with_real_floor(self._compressor(0), 10_000) == 10_000 assert _pressure_with_real_floor(object(), 10_000) == 10_000 + + def test_anchored_pressure_is_never_floored(self): + """A valid usage anchor is provider-exact and wins as-is. + + On MoA turns the anchor deliberately uses the pre-fold aggregator + usage while ``last_real_prompt_tokens`` holds the folded figure; + flooring the anchored value would re-add the advisor fan-out tokens + the anchor exists to exclude. Pin the wiring shape: the floor is + applied only on the ``else`` (rough fallback) branch. + """ + import inspect + from agent import conversation_loop + + src = inspect.getsource(conversation_loop.run_conversation) + i = src.index("if _anchored_pressure is not None:") + window = src[i : i + 400] + assert "request_pressure_tokens = _anchored_pressure" in window + assert "else:" in window + assert window.index("else:") < window.index("_pressure_with_real_floor(")