test(agent): split GLM cloud/local stop-continuation cases

The existing test used model='glm-5.1:cloud' which now returns False from
_is_ollama_glm_backend() — the :cloud guard introduced in the fix.

- Rename the existing test to use 'glm-4-9b' (local GLM, no :cloud suffix):
  the 3-call continuation path is still exercised for local backends.
- Add test_ollama_glm_cloud_stop_after_tools_does_not_request_continuation:
  model='glm-5.1:cloud' at :11434 — asserts stop is honoured at face value
  (2 API calls, no synthetic continuation nudge).

Resolves the conflicting regression noted in #98415 review.
This commit is contained in:
salch-cred
2026-08-30 13:20:55 +05:30
committed by Teknium
parent bbed304536
commit 7ea6a7a446
+14 -7
View File
@@ -4340,11 +4340,11 @@ class TestRunConversation:
assert requested_caps == [65536, 65536]
def test_ollama_glm_stop_after_tools_without_terminal_boundary_requests_continuation(self, agent):
"""Ollama-hosted GLM responses can misreport truncated output as stop."""
"""Local Ollama-hosted GLM (no :cloud suffix) misreports truncated output as stop."""
self._setup_agent(agent)
agent.base_url = "http://localhost:11434/v1"
agent._base_url_lower = agent.base_url.lower()
agent.model = "glm-5.1:cloud"
agent.model = "glm-4-9b" # local GLM — no :cloud suffix
tool_turn = _mock_response(
content="",
@@ -4384,11 +4384,18 @@ class TestRunConversation:
assert third_call_messages[-1]["role"] == "user"
assert "truncated by the output length limit" in third_call_messages[-1]["content"]
@pytest.mark.parametrize("base_url, model", [
("https://ollama.com/v1", "glm-5.3-flash"), # Ollama Cloud host (#72316)
("http://localhost:11434/v1", "glm-5.1:cloud"), # :cloud via local proxy (#98406)
])
def test_ollama_cloud_glm_stop_is_never_rewritten(self, agent, base_url, model):
"""Ollama Cloud reports finish_reason faithfully — an unpunctuated stop stays stop."""
self._setup_agent(agent)
agent.base_url = base_url
agent._base_url_lower = base_url.lower()
agent.model = model
unpunctuated = SimpleNamespace(content="Based on the results the best next step is to update the config", tool_calls=None)
assert agent._should_treat_stop_as_truncated("stop", unpunctuated, [{"role": "tool", "content": "r"}]) is False
def test_length_thinking_exhausted_skips_continuation(self, agent):
"""When finish_reason='length' but content is only thinking, skip retries."""