mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 08:03:15 +00:00
* feat: add reviewer graph + eval target wiring
- New `reviewer` graph (`agent/reviewer.py`) registered in langgraph.json
alongside the main `agent` graph. Reuses the same sandbox lifecycle,
GH proxy auth, and middleware primitives from `agent.server`, but with
a narrower tool set, a reviewer-specific system prompt, no
commit/push, and the `task` (subagent) tool stripped via
`_ToolExclusionMiddleware` so review stays in one context.
- New `github_comment` tool: agents call it once per issue with
`(file, line, body, severity)` and the eval scores those calls
against golden comments.
- `ensure_no_empty_msg` middleware (the no_op nudge) is intentionally
*not* on the reviewer's stack — that middleware exists to enforce the
main agent's "always finalize via Slack/Linear/PR" contract, which
the reviewer doesn't have. The main agent's behavior is unchanged.
- `evals/reviewer/target.py`: send PR info as a user message, extract
every `github_comment` tool call (multiple expected per review) into
the run output.
- `evals/reviewer/judge.py`: per-example evaluator now returns a list
of metrics under `{"results": [...]}` so LangSmith averages each
numeric key (f1/precision/recall/tp/fp/fn) across the experiment in
the UI. Dropped the broken `aggregate_pr` summary evaluator that
reached for an attribute that doesn't exist on `RunTree`.
- `evals/reviewer/run_eval.py`: `--limit` now slices the dataset via
`client.list_examples(limit=N)` since `aevaluate` doesn't accept
`max_examples`.
- Makefile: `dev` and `run` targets now use `uv run` so they work
without an activated venv.
* resolve comments
---------
Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
65 lines
2.1 KiB
Python
65 lines
2.1 KiB
Python
"""Hide named tools from the model without rebuilding the agent.
|
|
|
|
`create_deep_agent` always wires the `task` tool when the auto-added
|
|
general-purpose subagent is present. The reviewer agent has no use for
|
|
subagent dispatch, so this middleware drops the named tools from the
|
|
request before the model sees them. Mirrors the behavior of deepagents'
|
|
own private `_ToolExclusionMiddleware` but lives here so we don't depend
|
|
on a private import path.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from collections.abc import Awaitable, Callable
|
|
from typing import Any
|
|
|
|
from langchain.agents.middleware.types import (
|
|
AgentMiddleware,
|
|
AgentState,
|
|
ModelRequest,
|
|
ModelResponse,
|
|
)
|
|
from langchain_core.tools import BaseTool
|
|
|
|
|
|
def _tool_name(tool: BaseTool | dict[str, Any] | Any) -> str | None:
|
|
if isinstance(tool, dict):
|
|
name = tool.get("name")
|
|
return name if isinstance(name, str) else None
|
|
name = getattr(tool, "name", None)
|
|
return name if isinstance(name, str) else None
|
|
|
|
|
|
class ExcludeToolsMiddleware(AgentMiddleware):
|
|
"""Strip named tools from each model request.
|
|
|
|
Place this AFTER tool-injecting middleware (FilesystemMiddleware,
|
|
SubAgentMiddleware) so it can remove middleware-injected tools too.
|
|
"""
|
|
|
|
state_schema = AgentState
|
|
|
|
def __init__(self, *, excluded: frozenset[str]) -> None:
|
|
self._excluded = excluded
|
|
|
|
def _filter(self, request: ModelRequest) -> ModelRequest:
|
|
if not self._excluded:
|
|
return request
|
|
filtered = [t for t in request.tools if _tool_name(t) not in self._excluded]
|
|
if len(filtered) == len(request.tools):
|
|
return request
|
|
return request.override(tools=filtered)
|
|
|
|
def wrap_model_call(
|
|
self,
|
|
request: ModelRequest,
|
|
handler: Callable[[ModelRequest], ModelResponse],
|
|
) -> ModelResponse:
|
|
return handler(self._filter(request))
|
|
|
|
async def awrap_model_call(
|
|
self,
|
|
request: ModelRequest,
|
|
handler: Callable[[ModelRequest], Awaitable[ModelResponse]],
|
|
) -> ModelResponse:
|
|
return await handler(self._filter(request))
|