Phase 1 stabilization. Removes the four-copy prompt/agent-description drift surface and the silent router fallback. - models.py: hoist model IDs to module-level constants; add with_retries() helper (2 retries on Anthropic+OpenAI transient errors via with_retry). - agents.py: single AGENTS dict (model_fn, prompt, description) and a make_agent_node() factory that collapses six near-identical node functions. - graph.py: router prompt is generated from AGENTS; router_node uses with_structured_output(RouteDecision) and returns an explicit "unknown" route instead of the silent "researcher" fallback. New unknown_node wires to END. All LLM invocations go through with_retries. - state.py: add "unknown" to the route Literal. - run.py: --route-only now imports router_node from graph.py, killing the fourth prompt copy. - tests/: pytest golden-set (20 labelled tasks + size guard). Skips cleanly without ANTHROPIC_API_KEY or COMPOSIO_API_KEY. Validated 21/21 passing.
101 lines
4 KiB
Python
101 lines
4 KiB
Python
from langchain_core.messages import SystemMessage, HumanMessage
|
|
from state import OrchestratorState
|
|
from models import (
|
|
get_implementer,
|
|
get_reviewer,
|
|
get_researcher,
|
|
get_cross_reviewer,
|
|
get_scanner,
|
|
get_fast_coder,
|
|
with_retries,
|
|
)
|
|
|
|
IMPLEMENTER_PROMPT = """You are an implementation agent. You write clean, production-ready code.
|
|
Follow the spec exactly. No over-engineering, no unnecessary abstractions.
|
|
Return the complete implementation with file paths."""
|
|
|
|
REVIEWER_PROMPT = """You are a code review agent. Review code changes for correctness, security, and maintainability.
|
|
|
|
For each issue found, categorize it:
|
|
- BLOCK — Must fix before merge. Security vulnerabilities, data loss risks, broken logic.
|
|
- FIX — Should fix. Bugs, performance issues, missing error handling at boundaries.
|
|
- NIT — Optional. Style, naming, minor improvements.
|
|
- QUESTION — Needs clarification. Intent is unclear.
|
|
|
|
Start with a one-line summary: APPROVE, REQUEST CHANGES, or NEEDS DISCUSSION.
|
|
Then list findings grouped by category (BLOCK > FIX > NIT > QUESTION).
|
|
If the code is fine, say "No issues found." No praise or filler."""
|
|
|
|
RESEARCHER_PROMPT = """You are a research agent. Look up documentation, API references, and technical answers.
|
|
Be concise and cite sources. Return factual information, not opinions."""
|
|
|
|
CROSS_REVIEWER_PROMPT = """You are a cross-family code review agent. You provide an independent review perspective.
|
|
Review the provided code for bugs, security issues, and improvements.
|
|
Categorize findings as BLOCK, FIX, or NIT. Be concise.
|
|
Focus on issues that might be missed by the primary development team."""
|
|
|
|
SCANNER_PROMPT = """You are a large-context analysis agent. You analyze codebases, identify patterns,
|
|
check consistency across files, and find structural issues.
|
|
Summarize findings concisely with file references."""
|
|
|
|
FAST_CODER_PROMPT = """You are a fast coding agent for well-specified tasks.
|
|
Implement exactly what is asked. No extras, no refactoring beyond scope.
|
|
Return complete, working code."""
|
|
|
|
|
|
# Single source of truth for simple-pattern agents (system + human → result).
|
|
# Connector is handled separately in graph.py because it binds tools.
|
|
AGENTS = {
|
|
"implementer": {
|
|
"model_fn": get_implementer,
|
|
"prompt": IMPLEMENTER_PROMPT,
|
|
"description": "Write new code, add features, fix bugs. Use for any coding task with a clear spec.",
|
|
},
|
|
"reviewer": {
|
|
"model_fn": get_reviewer,
|
|
"prompt": REVIEWER_PROMPT,
|
|
"description": "Review code changes (diffs, PRs) for correctness, security, maintainability. Uses Claude.",
|
|
},
|
|
"researcher": {
|
|
"model_fn": get_researcher,
|
|
"prompt": RESEARCHER_PROMPT,
|
|
"description": "Look up documentation, API references, technical questions. Fast and cheap.",
|
|
},
|
|
"cross_reviewer": {
|
|
"model_fn": get_cross_reviewer,
|
|
"prompt": CROSS_REVIEWER_PROMPT,
|
|
"description": (
|
|
"Independent code review using a different AI model (GPT). Use when you want a "
|
|
"second opinion that catches different blind spots than Claude."
|
|
),
|
|
},
|
|
"scanner": {
|
|
"model_fn": get_scanner,
|
|
"prompt": SCANNER_PROMPT,
|
|
"description": (
|
|
"Analyze large codebases for patterns, consistency, structural issues. "
|
|
"Uses Gemini's large context window."
|
|
),
|
|
},
|
|
"fast_coder": {
|
|
"model_fn": get_fast_coder,
|
|
"prompt": FAST_CODER_PROMPT,
|
|
"description": "Quick, bounded coding for crystal-clear specs. Uses DeepSeek. Best for small, well-defined tasks.",
|
|
},
|
|
}
|
|
|
|
|
|
def make_agent_node(label: str):
|
|
cfg = AGENTS[label]
|
|
|
|
def node(state: OrchestratorState) -> dict:
|
|
llm = with_retries(cfg["model_fn"]())
|
|
response = llm.invoke(
|
|
[
|
|
SystemMessage(content=cfg["prompt"]),
|
|
HumanMessage(content=state["task"]),
|
|
]
|
|
)
|
|
return {"result": response.content, "messages": [response]}
|
|
|
|
return node
|