fix(agent): never floor an anchored pressure figure; keep the estimator total on lone surrogates

Follow-ups from review of the two salvaged #87490 commits:

- _pressure_with_real_floor now applies only on the rough fallback branch.
  A valid usage anchor is provider-exact and wins as-is: on MoA turns the
  anchor deliberately uses the pre-fold aggregator usage while
  last_real_prompt_tokens holds the folded figure, so flooring the anchored
  value would re-add fan-out tokens the anchor exists to exclude. Docstring
  rewritten to describe the real path split (anchor since d3a1c46510).
- estimate_tokens_rough: encode with errors="replace". main's estimator
  never raised; text.encode() on a lone surrogate (routine in tool output,
  see message_sanitization) raised UnicodeEncodeError and would abort a
  turn where main produced a slightly-off number.
- Record the cl100k/o200k/Qwen2.5 calibration for the bytes/4 rule.
- tests: accented Latin within +10% of the ASCII rule; mixed Cyrillic/ASCII
  counts ASCII at one byte; lone surrogates don't raise; anchored pressure
  is never floored (wiring shape).
This commit is contained in:
kshitijk4poor
2026-09-03 02:48:58 +05:30
committed by kshitij
parent f07b6ff426
commit a1d5a976b3
4 changed files with 79 additions and 13 deletions
+20 -11
View File
@@ -610,15 +610,23 @@ def _image_error_max_dimension(error: Exception) -> Optional[int]:
def _pressure_with_real_floor(compressor: Any, rough_tokens: int) -> int:
"""Floor the rough pre-API pressure estimate at the last REAL prompt size.
"""Floor the ROUGH pre-API pressure estimate at the last REAL prompt size.
The chars/4 rough estimate under-counts Cyrillic (and other non-ASCII
scripts) by up to ~2x, so a session can sit at the provider's real context
ceiling while the rough figure stays under the compaction threshold — on
silent-clip providers (ollama /v1) that is a truncation death spiral the
reactive overflow handler never sees (observed live: real prompts
64,842→64,995 against a 55,705 threshold). The provider's own last
reported prompt_tokens is authoritative; never let the pressure figure
Applied only on the fallback path -- when ``anchored_context_tokens`` has
no valid anchor (first request, transcript rewritten under the anchor,
provider never reported usage). A valid anchor is provider-exact and is
used as-is; in particular on MoA turns the anchor deliberately uses the
pre-fold aggregator usage while ``last_real_prompt_tokens`` holds the
folded figure, so flooring an anchored value would re-add fan-out tokens
the anchor exists to exclude.
On the rough path, non-ASCII text (Cyrillic, Greek, Polish, ...)
under-counts by up to ~2x, so a session can sit at the provider's real
context ceiling while the rough figure stays under the compaction
threshold -- on silent-clip providers (ollama /v1) that is a truncation
death spiral the reactive overflow handler never sees (observed live:
real prompts 64,842->64,995 against a 55,705 threshold). The provider's
last reported prompt_tokens is authoritative; never let the rough figure
fall below it. Skipped for exactly one turn after a compaction, when
last_real_prompt_tokens still holds the stale pre-compression value
(#36718's awaiting_real_usage_after_compression window).
@@ -2904,9 +2912,10 @@ def run_conversation(
)
if _anchored_pressure is not None:
request_pressure_tokens = _anchored_pressure
request_pressure_tokens = _pressure_with_real_floor(
agent.context_compressor, request_pressure_tokens
)
else:
request_pressure_tokens = _pressure_with_real_floor(
agent.context_compressor, request_pressure_tokens
)
total_chars = approx_tokens * 4
# Stash this request's rough estimate so update_from_response() can
# pair it with the provider's real prompt count — the (rough, real)
+11 -2
View File
@@ -3662,10 +3662,19 @@ def estimate_tokens_rough(text: str) -> int:
# token) where chars/4 under-counted them ~2x and let sessions ride
# the provider's context ceiling below the compaction threshold.
# ASCII spans inside mixed text still count at 1 byte each.
return (len(text.encode("utf-8")) + 3) // 4
#
# Calibrated against cl100k/o200k/Qwen2.5 (estimate / mean real):
# Russian 0.67->1.24, Ukrainian 0.55->1.03, Arabic 0.53->0.96,
# Hindi 0.34->0.90, Greek 0.37->0.68, Polish 0.63->0.69; accented
# Latin barely moves (French 1.02->1.03, German 0.99->1.02,
# Spanish 1.04->1.07) because only the accented chars widen.
# Pure-ASCII prose already over-counts at ~1.4 on the same rule.
# errors="replace": lone surrogates (routine in tool output; see
# message_sanitization) must not turn an estimate into a raise.
return (len(text.encode("utf-8", "replace")) + 3) // 4
# Mixed CJK + other: dense chars stay ~1 token each; the sparse
# remainder is byte-counted for the same corrective.
return dense + ((len(stripped.encode("utf-8")) + 3) // 4)
return dense + ((len(stripped.encode("utf-8", "replace")) + 3) // 4)
def estimate_messages_tokens_rough(
+29
View File
@@ -88,3 +88,32 @@ def test_cyrillic_counts_by_utf8_bytes():
assert estimate_tokens_rough("русский текст") == 7
# Pure ASCII unchanged.
assert estimate_tokens_rough("a" * 400) == 100
def test_accented_latin_is_not_inflated_by_byte_counting():
# Byte-counting must not punish Western-European text: only the accented
# chars are 2 bytes, so the estimate moves by a few percent, not 2x.
from agent.model_metadata import estimate_tokens_rough
fr = "La compression du contexte permet aux longues sessions de rester dans la fenêtre du fournisseur sans perdre le fil de la tâche."
ascii_rule = (len(fr) + 3) // 4
est = estimate_tokens_rough(fr)
assert ascii_rule <= est <= int(ascii_rule * 1.10), (ascii_rule, est)
def test_mixed_cyrillic_and_ascii_code_counts_ascii_at_one_byte():
from agent.model_metadata import estimate_tokens_rough
code = "def compress(ctx):\n # Сжимаем контекст\n return summarize(ctx)\n"
ascii_part = "def compress(ctx):\n # \n return summarize(ctx)\n"
cyr = "Сжимаем контекст"
expected = (len(ascii_part.encode()) + len(cyr.encode()) + 3) // 4
assert estimate_tokens_rough(code) == expected
# and strictly more than the old chars/4 rule for the same text
assert estimate_tokens_rough(code) > (len(code) + 3) // 4
def test_lone_surrogates_do_not_raise():
# main's estimator was total (len/regex never raise); byte-counting must
# stay total too — tool output routinely carries unpaired surrogates.
from agent.model_metadata import estimate_tokens_rough
assert estimate_tokens_rough("abc\ud800def") >= 2
assert estimate_tokens_rough("漢字\udfff") >= 2
@@ -292,3 +292,22 @@ class TestPressureRealFloor:
from agent.conversation_loop import _pressure_with_real_floor
assert _pressure_with_real_floor(self._compressor(0), 10_000) == 10_000
assert _pressure_with_real_floor(object(), 10_000) == 10_000
def test_anchored_pressure_is_never_floored(self):
"""A valid usage anchor is provider-exact and wins as-is.
On MoA turns the anchor deliberately uses the pre-fold aggregator
usage while ``last_real_prompt_tokens`` holds the folded figure;
flooring the anchored value would re-add the advisor fan-out tokens
the anchor exists to exclude. Pin the wiring shape: the floor is
applied only on the ``else`` (rough fallback) branch.
"""
import inspect
from agent import conversation_loop
src = inspect.getsource(conversation_loop.run_conversation)
i = src.index("if _anchored_pressure is not None:")
window = src[i : i + 400]
assert "request_pressure_tokens = _anchored_pressure" in window
assert "else:" in window
assert window.index("else:") < window.index("_pressure_with_real_floor(")