From 7ea6a7a4469097732e64f57f6f6fefd9af0dd6fd Mon Sep 17 00:00:00 2001 From: salch-cred Date: Sun, 30 Aug 2026 13:20:55 +0530 Subject: [PATCH] test(agent): split GLM cloud/local stop-continuation cases MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The existing test used model='glm-5.1:cloud' which now returns False from _is_ollama_glm_backend() — the :cloud guard introduced in the fix. - Rename the existing test to use 'glm-4-9b' (local GLM, no :cloud suffix): the 3-call continuation path is still exercised for local backends. - Add test_ollama_glm_cloud_stop_after_tools_does_not_request_continuation: model='glm-5.1:cloud' at :11434 — asserts stop is honoured at face value (2 API calls, no synthetic continuation nudge). Resolves the conflicting regression noted in #98415 review. --- tests/run_agent/test_run_agent.py | 21 ++++++++++++++------- 1 file changed, 14 insertions(+), 7 deletions(-) diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py index 604364d269..069bd55d69 100644 --- a/tests/run_agent/test_run_agent.py +++ b/tests/run_agent/test_run_agent.py @@ -4340,11 +4340,11 @@ class TestRunConversation: assert requested_caps == [65536, 65536] def test_ollama_glm_stop_after_tools_without_terminal_boundary_requests_continuation(self, agent): - """Ollama-hosted GLM responses can misreport truncated output as stop.""" + """Local Ollama-hosted GLM (no :cloud suffix) misreports truncated output as stop.""" self._setup_agent(agent) agent.base_url = "http://localhost:11434/v1" agent._base_url_lower = agent.base_url.lower() - agent.model = "glm-5.1:cloud" + agent.model = "glm-4-9b" # local GLM — no :cloud suffix tool_turn = _mock_response( content="", @@ -4384,11 +4384,18 @@ class TestRunConversation: assert third_call_messages[-1]["role"] == "user" assert "truncated by the output length limit" in third_call_messages[-1]["content"] - - - - - + @pytest.mark.parametrize("base_url, model", [ + ("https://ollama.com/v1", "glm-5.3-flash"), # Ollama Cloud host (#72316) + ("http://localhost:11434/v1", "glm-5.1:cloud"), # :cloud via local proxy (#98406) + ]) + def test_ollama_cloud_glm_stop_is_never_rewritten(self, agent, base_url, model): + """Ollama Cloud reports finish_reason faithfully — an unpunctuated stop stays stop.""" + self._setup_agent(agent) + agent.base_url = base_url + agent._base_url_lower = base_url.lower() + agent.model = model + unpunctuated = SimpleNamespace(content="Based on the results the best next step is to update the config", tool_calls=None) + assert agent._should_treat_stop_as_truncated("stop", unpunctuated, [{"role": "tool", "content": "r"}]) is False def test_length_thinking_exhausted_skips_continuation(self, agent): """When finish_reason='length' but content is only thinking, skip retries."""