From 3ff9a43ca3ff2ea90f9415647b5a8158cc8dbe7c Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Wed, 24 Jun 2026 15:18:28 -0400 Subject: [PATCH] fix(agent-team): clarifier turn headroom (max_turns=4) so it can finish its JSON MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The clarifier (ClaudeClarifier._turn) called claude_invoke with no max_turns, inheriting the single-shot default (1). When the model's one turn did not terminate in a final result the SDK raised 'Reached maximum number of turns (1)' and, with no salvageable text, the call failed and crashed the clarify node — leaving the task wedged at clarify with NO question posted to Slack (the human never sees a clarifier prompt). Observed live on the R720. Same single-shot flake the planner hit and fixed in PR #58 (_PLANNER_MAX_TURNS=4); the clarifier never got the headroom. Give it the same: pass max_turns=4 (tools stay off — still a fast reasoning->JSON completion). - clarifier_llm.py: _turn passes max_turns=_CLARIFIER_MAX_TURNS (=4). - tests: clarifier passes max_turns headroom to the invoke seam. Full suite 1505 passed; ruff clean. --- agent-team/agent_team/nodes/clarifier_llm.py | 17 ++++++++++++++++- agent-team/tests/test_clarifier_llm.py | 12 ++++++++++++ 2 files changed, 28 insertions(+), 1 deletion(-) diff --git a/agent-team/agent_team/nodes/clarifier_llm.py b/agent-team/agent_team/nodes/clarifier_llm.py index 68b5fd4..e5440ab 100644 --- a/agent-team/agent_team/nodes/clarifier_llm.py +++ b/agent-team/agent_team/nodes/clarifier_llm.py @@ -61,6 +61,16 @@ __all__ = [ # clarifier loop's injected callables, etc.). ClaudeInvoke = Callable[..., ClaudeResult] +# The clarifier is a single-shot reasoning→JSON completion (confidence + +# question-set), but the invoker's single-shot default (max_turns=1) is flaky: +# when the model's one turn does not terminate in a final result it raises +# "Reached maximum number of turns (1)", and with no salvageable text the call +# fails and crashes the clarify node (leaving the task wedged at clarify with no +# question posted). The planner hit the same flake and was given headroom in PR +# #58; the clarifier needs the same. A few turns let the model FINISH its JSON; +# tools stay OFF so it remains a fast, deterministic completion. +_CLARIFIER_MAX_TURNS = 4 + # Used when the model is below the confidence bar but supplied no usable # question-set. The loop must always have something to ask rather than spin or # falsely advance, so we substitute a generic clarifier prompt. @@ -202,7 +212,12 @@ class ClaudeClarifier: return self._cache prompt = self._build_prompt(qa_history, state) - result = self._invoke(prompt, model=self._model, config=self._config) + result = self._invoke( + prompt, + model=self._model, + config=self._config, + max_turns=_CLARIFIER_MAX_TURNS, + ) parsed = self._parse(getattr(result, "text", "")) self._cache_key = key diff --git a/agent-team/tests/test_clarifier_llm.py b/agent-team/tests/test_clarifier_llm.py index 1448709..1379d77 100644 --- a/agent-team/tests/test_clarifier_llm.py +++ b/agent-team/tests/test_clarifier_llm.py @@ -89,6 +89,18 @@ def test_high_confidence_parsed() -> None: assert clar.assess_confidence([], _state()) == 0.99 +def test_clarifier_passes_max_turns_headroom_to_invoke_seam() -> None: + # The single-shot Claude default (1 turn) is flaky: it crashes the clarify + # node with "Reached maximum number of turns (1)" and leaves the task wedged + # with no question posted. The clarifier asks for headroom so the model can + # FINISH its JSON (mirrors the planner fix, PR #58). + fake = _FakeInvoke(_json(0.99, [])) + clar = ClaudeClarifier(invoke=fake) + clar.assess_confidence([], _state()) + assert fake.calls[0]["kw"].get("max_turns") == 4 + assert fake.calls[0]["kw"]["max_turns"] > 1 + + def test_single_call_per_turn_memoized() -> None: fake = _FakeInvoke(_json(0.99, [])) clar = ClaudeClarifier(invoke=fake)