This repository has been archived on 2026-08-04. You can view files and clone it, but cannot push or open issues or pull requests.
orchestrator/agent-team/agent_team/nodes/clarifier_llm.py
Adam Moussa 270ce93b2a feat(agent-team): bind P1 clarifier to real Claude via subscription-OAuth invoker
Adds the billing-seam invoker (claude_agent_sdk subscription-OAuth, deferred import, API/Bedrock paths) and the Claude-backed clarifier callables (ConfidenceAssessor/QuestionGenerator, one call/turn memoized on (thread_id,len,content-hash), fail-safe to 0.0 so garbage never clears the 98% human gate).
2026-06-18 12:56:42 -04:00

466 lines
19 KiB
Python

"""Claude-backed clarifier callables — the real §3.3 / §7.1 P1 bindings.
:mod:`agent_team.nodes.clarifier` owns the *loop* (the LangGraph
``interrupt()``/resume 98% gate) but deliberately injects the two reasoning
seams so the loop stays pure and testable:
* ``ConfidenceAssessor = Callable[[Sequence[object], PipelineState], float]``
* ``QuestionGenerator = Callable[[Sequence[object], PipelineState], list[str]]``
This module supplies the **real, Claude-backed** implementations of those two
callables. It calls Claude only through the committed
:func:`agent_team.billing.claude_invoke` seam (§3.1) — never a raw SDK — so the
billing-mode hygiene and the budget ledger stay in one place.
The naive binding is wasteful: the clarifier loop calls ``assess_confidence``
and then ``generate_questions`` separately on the same turn, so two independent
implementations would make **two** Claude calls per turn for what is really one
reasoning step. :class:`ClaudeClarifier` instead makes **one** Claude call per
turn and serves both methods from the memoized result. The memo is keyed on the
Q&A history length, so a new answer (history grows) recomputes, while the
back-to-back assess/generate pair within one turn reuses the same call.
Defensive parsing is a hard requirement here because the model output is
UNTRUSTED and this is the **human gate** (§3.3): a parse failure must *never*
clear the gate. The parser fails SAFE — a missing/garbled confidence defaults to
``0.0`` (so the loop keeps asking rather than falsely advancing to planning),
and a missing question-set below threshold falls back to a single generic
clarifying question (so the loop still has something to ask).
"""
from __future__ import annotations
import hashlib
import json
import re
from collections.abc import Callable, Sequence
from typing import Any
from agent_team.billing import ClaudeResult, claude_invoke
from agent_team.nodes.clarifier import (
DEFAULT_CONFIDENCE_THRESHOLD,
ConfidenceAssessor,
QuestionGenerator,
)
from agent_team.task_model import PipelineState
__all__ = [
"FALLBACK_QUESTION",
"ClaudeClarifier",
"build_claude_clarifier_callables",
]
# The signature the billing seam exposes: ``claude_invoke(prompt, *, mode=None,
# config=None, **kw) -> ClaudeResult``. Injected so tests pass a fake, mirroring
# the injection pattern used across this codebase (billing.set_invoker, the
# clarifier loop's injected callables, etc.).
ClaudeInvoke = Callable[..., ClaudeResult]
# Used when the model is below the confidence bar but supplied no usable
# question-set. The loop must always have something to ask rather than spin or
# falsely advance, so we substitute a generic clarifier prompt.
FALLBACK_QUESTION = (
"Could you share more about the goal, scope, and constraints of this task "
"so I can be sure I understand it well enough to plan?"
)
# Default system framing handed to Claude. Kept as a module constant so callers
# can override via the ``system`` constructor hook without forking the class.
_DEFAULT_SYSTEM = (
"You are the CLARIFIER stage of an agentic SDLC pipeline and the human "
"gate before any planning happens. Your job is to decide whether the "
"requirement is understood well enough to plan, drawing conceptually on "
"the repo, prior memory, and the engineering handbook. Be rigorous: only "
"report high confidence when the goal, scope, and constraints are "
"genuinely unambiguous."
)
def _turn_cache_key(
qa_history: Sequence[object], state: PipelineState
) -> tuple[str, int, str]:
"""Build the task-scoped memo key for one clarifier turn.
Binds the ``thread_id`` (task isolation), the history length (turn index),
and a content hash of the Q&A so far. The thread id is the load-bearing
part: one :class:`ClaudeClarifier` instance is shared by the long-lived
graph node across every task, so keying on length alone would let one
task's cached confidence satisfy another task's gate with no model call.
The content hash is belt-and-suspenders so an in-place edit of the same-
length history (should one ever occur) also invalidates the memo.
"""
thread_id = str(state.get("thread_id", "") if isinstance(state, dict) else "")
try:
digest_src = json.dumps(list(qa_history), sort_keys=True, default=repr)
except (TypeError, ValueError):
digest_src = repr(list(qa_history))
content_hash = hashlib.sha1(digest_src.encode("utf-8")).hexdigest()
return (thread_id, len(qa_history), content_hash)
class ClaudeClarifier:
"""One Claude call per turn, serving both clarifier callables (§3.3, §7.1 P1).
Construct with an optional ``invoke`` callable (defaults to
:func:`agent_team.billing.claude_invoke`) so tests inject a fake and the
real wiring goes through the billing seam. ``model`` / ``config`` are passed
through to the invoker, and ``system`` overrides the prompt framing.
The single call per turn is memoized on a task-scoped key
(``thread_id`` + history length + content hash, see :func:`_turn_cache_key`):
calling :meth:`assess_confidence` then :meth:`generate_questions` for the
same turn of the same task reuses one Claude call; appending an answer (the
history grows) or a different task entering the shared node invalidates the
memo and the next assess triggers a fresh call. The thread-scoping is what
stops one task's cached confidence from clearing another task's human gate.
:meth:`assess_confidence` and :meth:`generate_questions` are bound methods
that match :data:`~agent_team.nodes.clarifier.ConfidenceAssessor` and
:data:`~agent_team.nodes.clarifier.QuestionGenerator` exactly, so they wire
straight into :func:`~agent_team.nodes.clarifier.make_clarifier_node`.
"""
def __init__(
self,
*,
invoke: ClaudeInvoke | None = None,
model: str | None = None,
config: Any = None,
system: str = _DEFAULT_SYSTEM,
confidence_threshold: float = DEFAULT_CONFIDENCE_THRESHOLD,
) -> None:
self._invoke: ClaudeInvoke = invoke if invoke is not None else claude_invoke
self._model = model
self._config = config
self._system = system
self._confidence_threshold = confidence_threshold
# Memo of the single per-turn call. The key is task-scoped, NOT just the
# history length: one ClaudeClarifier instance serves every task through
# the long-lived graph node, so a key of len(qa_history) alone would let
# one task's cached high confidence clear ANOTHER task's human gate with
# no Claude call (a fail-OPEN cross-task collision). The key therefore
# binds (thread_id, history-length, content-hash) so the memo isolates
# per task/thread and still recomputes when the Q&A changes.
self._cache_key: tuple[str, int, str] | None = None
self._cache: dict[str, Any] | None = None
# ------------------------------------------------------------------ #
# Public callables — exact ConfidenceAssessor / QuestionGenerator types.
# ------------------------------------------------------------------ #
def assess_confidence(
self, qa_history: Sequence[object], state: PipelineState
) -> float:
"""Return the current 0..1 confidence the requirement is understood.
Matches :data:`~agent_team.nodes.clarifier.ConfidenceAssessor`. Serves
the memoized per-turn Claude call; fails SAFE to ``0.0`` on any parse
trouble so a garbled response never clears the human gate.
"""
return float(self._turn(qa_history, state)["confidence"])
def generate_questions(
self, qa_history: Sequence[object], state: PipelineState
) -> list[str]:
"""Return the next ordered question-set.
Matches :data:`~agent_team.nodes.clarifier.QuestionGenerator`. Reuses
the same memoized call as :meth:`assess_confidence` for this turn, and
always returns a non-empty list (the loop must have something to ask).
"""
return list(self._turn(qa_history, state)["questions"])
# ------------------------------------------------------------------ #
# Internals: the single per-turn call + memo.
# ------------------------------------------------------------------ #
def _turn(
self, qa_history: Sequence[object], state: PipelineState
) -> dict[str, Any]:
"""Return the parsed result for this turn, making at most one Claude call.
Memoized on ``(thread_id, len(qa_history), content-hash)``: the
assess/generate pair within one turn of one task shares a call; once an
answer is appended (history grows) or a different task/thread enters the
shared node, the key changes and a fresh call is made. Keying on the
thread id is what prevents one task's cached confidence from clearing
another task's human gate (the fail-OPEN collision the review caught).
"""
key = _turn_cache_key(qa_history, state)
if self._cache_key == key and self._cache is not None:
return self._cache
prompt = self._build_prompt(qa_history, state)
result = self._invoke(prompt, model=self._model, config=self._config)
parsed = self._parse(getattr(result, "text", ""))
self._cache_key = key
self._cache = parsed
return parsed
def _build_prompt(self, qa_history: Sequence[object], state: PipelineState) -> str:
"""Assemble the clarifier prompt from the Q&A history and task state.
Pure string assembly over the graph state (§3.3) — no I/O — so the
prompt shape is directly unit-testable.
"""
description = _task_description(state)
repo = _state_field(state, "repo")
context = _state_field(state, "context")
qa = _format_qa_history(qa_history)
threshold_pct = int(round(self._confidence_threshold * 100))
sections: list[str] = [
self._system,
"",
"## Task",
description or "(no task description provided)",
]
if repo:
sections += ["", "## Repository", repo]
if context:
sections += ["", "## Additional context", context]
sections += [
"",
"## Clarifier Q&A so far (oldest first)",
qa or "(no questions answered yet)",
"",
"## Your job",
(
f"Decide whether you are at least {threshold_pct}% confident the "
"requirement is understood well enough to plan. If you are NOT, "
"produce the next ordered set of clarifying questions to ask the "
"human. Ask only what is genuinely needed; order them most "
"important first."
),
"",
"## Output format",
(
"Respond with ONLY a strict JSON object and no prose outside it, "
'with keys: "confidence" (a float in [0, 1]), "questions" (a list '
"of strings; empty only when you are confident enough to plan), "
'and "rationale" (a short string). Example: '
'{"confidence": 0.42, "questions": ["..."], "rationale": "..."}'
),
]
return "\n".join(sections)
def _parse(self, text: str) -> dict[str, Any]:
"""Parse the UNTRUSTED model reply into ``{confidence, questions, rationale}``.
Fails SAFE at every step (§3.3 human gate):
* confidence missing/unparseable -> ``0.0`` (keep asking, never clear
the gate on a garbled reply);
* confidence out of range -> clamped into ``[0, 1]``;
* questions missing/empty while below threshold -> a single generic
fallback question so the loop always has something to ask.
A parse error is swallowed into the fail-safe default rather than
raised, so a bad reply degrades to "ask again", never to "advance".
"""
data = _extract_json_object(text)
confidence = _coerce_confidence(data.get("confidence") if data else None)
questions = _coerce_questions(data.get("questions") if data else None)
rationale = ""
if data is not None:
raw_rationale = data.get("rationale")
if isinstance(raw_rationale, str):
rationale = raw_rationale.strip()
if not questions and confidence < self._confidence_threshold:
# Below the bar but no usable question-set: substitute a generic
# clarifier so the loop still asks rather than spinning or advancing.
questions = [FALLBACK_QUESTION]
return {
"confidence": confidence,
"questions": questions,
"rationale": rationale,
}
def build_claude_clarifier_callables(
*,
invoke: ClaudeInvoke | None = None,
model: str | None = None,
config: Any = None,
system: str = _DEFAULT_SYSTEM,
confidence_threshold: float = DEFAULT_CONFIDENCE_THRESHOLD,
) -> tuple[ConfidenceAssessor, QuestionGenerator]:
"""Build the ``(assess_confidence, generate_questions)`` pair for wiring.
Returns the two bound methods of a single shared :class:`ClaudeClarifier`,
ready to hand straight to
:func:`~agent_team.nodes.clarifier.make_clarifier_node`. Because both
callables share one instance, they share the per-turn memo, so the loop
makes one Claude call per turn rather than two.
"""
clarifier = ClaudeClarifier(
invoke=invoke,
model=model,
config=config,
system=system,
confidence_threshold=confidence_threshold,
)
return clarifier.assess_confidence, clarifier.generate_questions
# --------------------------------------------------------------------------- #
# Module-level helpers (pure; no I/O).
# --------------------------------------------------------------------------- #
def _state_field(state: PipelineState, key: str) -> str:
"""Pull a string field from the (untyped-extra) graph state, defensively."""
value = state.get(key) # type: ignore[call-overload]
if isinstance(value, str) and value.strip():
return value.strip()
return ""
def _task_description(state: PipelineState) -> str:
"""Pull the task description out of the graph state (mirrors planner.py).
Looks in the conventional places (the ``plan`` dict, then a top-level
``task``/``description`` key) and falls back to an empty string so a
malformed state surfaces as an empty prompt section, never a ``KeyError``.
"""
plan = state.get("plan") or {}
if isinstance(plan, dict):
desc = plan.get("task") or plan.get("description")
if isinstance(desc, str) and desc.strip():
return desc.strip()
for key in ("task", "description"):
desc = _state_field(state, key)
if desc:
return desc
return ""
def _format_qa_history(qa_history: Sequence[object]) -> str:
"""Render the clarifier Q&A history (oldest first) into prompt text.
Each entry may be a ``{"question": ..., "answer": ...}`` mapping or a plain
string (the raw resume value the loop appends); both are handled so this
does not couple to a single record shape.
"""
lines: list[str] = []
for idx, entry in enumerate(qa_history, start=1):
if isinstance(entry, dict):
question = str(entry.get("question", "")).strip()
answer = str(entry.get("answer", "")).strip()
if question or answer:
lines.append(f"{idx}. Q: {question}\n A: {answer}")
else:
text = str(entry).strip()
if text:
lines.append(f"{idx}. {text}")
return "\n".join(lines)
# A fenced ```json ... ``` block, if the model wrapped its JSON in Markdown.
_FENCE_RE = re.compile(
r"```(?:json)?\s*\n?(?P<body>.*?)\n?\s*```",
flags=re.DOTALL | re.IGNORECASE,
)
def _extract_json_object(text: str) -> dict[str, Any] | None:
"""Extract a JSON object from UNTRUSTED model output, or ``None``.
Tolerates the common ways a model deviates from "JSON only": a leading
apology or trailing prose, and ```json fences. Tries, in order, the whole
string, the contents of a fenced block, then the first ``{...}`` span found
by brace matching. Returns ``None`` (never raises) when nothing parses to a
JSON object, so the caller can fail SAFE.
"""
if not isinstance(text, str) or not text.strip():
return None
candidates: list[str] = [text.strip()]
fence = _FENCE_RE.search(text)
if fence:
candidates.append(fence.group("body").strip())
span = _first_brace_span(text)
if span is not None:
candidates.append(span)
for candidate in candidates:
if not candidate:
continue
try:
parsed = json.loads(candidate)
except (json.JSONDecodeError, ValueError):
continue
if isinstance(parsed, dict):
return parsed
return None
def _first_brace_span(text: str) -> str | None:
"""Return the first balanced ``{...}`` span in ``text`` (string-aware)."""
start = text.find("{")
if start == -1:
return None
depth = 0
in_string = False
escaped = False
for idx in range(start, len(text)):
ch = text[idx]
if in_string:
if escaped:
escaped = False
elif ch == "\\":
escaped = True
elif ch == '"':
in_string = False
continue
if ch == '"':
in_string = True
elif ch == "{":
depth += 1
elif ch == "}":
depth -= 1
if depth == 0:
return text[start : idx + 1]
return None
def _coerce_confidence(value: Any) -> float:
"""Coerce the model's confidence into a clamped ``[0, 1]`` float.
Missing or unparseable -> ``0.0`` (fail SAFE: keep asking, never clear the
gate). Out-of-range values are clamped rather than rejected.
"""
try:
confidence = float(value)
except (TypeError, ValueError):
return 0.0
if confidence != confidence: # NaN guard
return 0.0
if confidence < 0.0:
return 0.0
if confidence > 1.0:
return 1.0
return confidence
def _coerce_questions(value: Any) -> list[str]:
"""Coerce the model's question-set into a clean list of non-empty strings.
Anything that is not a list of usable strings collapses to an empty list,
which the parser then fills with the generic fallback when below threshold.
"""
if not isinstance(value, list):
return []
questions: list[str] = []
for item in value:
if isinstance(item, str):
text = item.strip()
if text:
questions.append(text)
return questions