mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 13:53:15 +00:00
fix: sanitize malformed Anthropic thinking blocks (#1357)
Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> Co-authored-by: Johannes du Plessis <51395795+johannes117@users.noreply.github.com>
This commit is contained in:
parent
eb92947806
commit
65acc4c9d3
6 changed files with 152 additions and 2 deletions
|
|
@ -68,9 +68,10 @@ Configured in `agent/server.py:get_agent`, runs around every model call (in this
|
|||
6. `ensure_no_empty_msg` — guards against empty assistant messages that some providers reject.
|
||||
7. `notify_step_limit_reached` — after-agent hook that posts a Slack reply when the agent hits the step limit, so the user gets a clear signal instead of silence.
|
||||
8. `SandboxCircuitBreakerMiddleware` — trips the agent out of repeated sandbox failures instead of looping.
|
||||
9. `ModelFallbackMiddleware` (optional, last) — added only when `LLM_FALLBACK_MODEL_ID` or the per-model default fallback differs from the primary model.
|
||||
9. `ModelFallbackMiddleware` (optional) — added only when `LLM_FALLBACK_MODEL_ID` or the per-model default fallback differs from the primary model.
|
||||
10. `SanitizeThinkingBlocksMiddleware` — strips malformed empty Anthropic thinking blocks immediately before provider calls.
|
||||
|
||||
Other middleware exists in `agent/middleware/` (`ExcludeToolsMiddleware`) but isn't wired into the default agent. The reviewer uses a leaner stack: `SanitizeToolInputsMiddleware`, `ModelCallLimitMiddleware`, `ToolErrorMiddleware`, `SlackAssistantStatusMiddleware`.
|
||||
Other middleware exists in `agent/middleware/` (`ExcludeToolsMiddleware`) but isn't wired into the default agent. The reviewer uses a leaner stack: `SanitizeToolInputsMiddleware`, `ModelCallLimitMiddleware`, `ToolErrorMiddleware`, `SlackAssistantStatusMiddleware`, `SanitizeThinkingBlocksMiddleware`.
|
||||
|
||||
There is intentionally no after-agent safety net that opens a PR for the agent. The agent itself is responsible for committing, pushing, opening/updating the draft PR, and replying in the source channel — all via `GH_TOKEN=dummy gh` and `slack_thread_reply` / `linear_comment`.
|
||||
|
||||
|
|
|
|||
|
|
@ -5,12 +5,14 @@ from .model_fallback import ModelFallbackMiddleware
|
|||
from .notify_step_limit import notify_step_limit_reached
|
||||
from .refresh_slack_status import SlackAssistantStatusMiddleware
|
||||
from .sandbox_circuit_breaker import SandboxCircuitBreakerMiddleware
|
||||
from .sanitize_thinking_blocks import SanitizeThinkingBlocksMiddleware
|
||||
from .sanitize_tool_inputs import SanitizeToolInputsMiddleware
|
||||
from .tool_error_handler import ToolErrorMiddleware
|
||||
|
||||
__all__ = [
|
||||
"ExcludeToolsMiddleware",
|
||||
"ModelFallbackMiddleware",
|
||||
"SanitizeThinkingBlocksMiddleware",
|
||||
"SanitizeToolInputsMiddleware",
|
||||
"ToolErrorMiddleware",
|
||||
"SandboxCircuitBreakerMiddleware",
|
||||
|
|
|
|||
67
agent/middleware/sanitize_thinking_blocks.py
Normal file
67
agent/middleware/sanitize_thinking_blocks.py
Normal file
|
|
@ -0,0 +1,67 @@
|
|||
"""Middleware that removes malformed Anthropic thinking blocks before model calls."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Awaitable, Callable
|
||||
from typing import Any
|
||||
|
||||
from langchain.agents.middleware import AgentMiddleware
|
||||
from langchain.agents.middleware.types import ModelCallResult, ModelRequest, ModelResponse
|
||||
from langchain_anthropic import ChatAnthropic
|
||||
from langchain_core.messages import AIMessage
|
||||
|
||||
|
||||
def _is_chat_anthropic(model: object) -> bool:
|
||||
seen: set[int] = set()
|
||||
current = model
|
||||
for _ in range(10):
|
||||
if isinstance(current, ChatAnthropic):
|
||||
return True
|
||||
current_id = id(current)
|
||||
if current_id in seen:
|
||||
return False
|
||||
seen.add(current_id)
|
||||
bound = getattr(current, "bound", None)
|
||||
if bound is None or bound is current:
|
||||
return False
|
||||
current = bound
|
||||
return False
|
||||
|
||||
|
||||
def _sanitize_messages(messages: list[Any]) -> None:
|
||||
for message in messages:
|
||||
if not isinstance(message, AIMessage) or not isinstance(message.content, list):
|
||||
continue
|
||||
content = [
|
||||
block
|
||||
for block in message.content
|
||||
if not (
|
||||
isinstance(block, dict)
|
||||
and block.get("type") == "thinking"
|
||||
and not block.get("thinking")
|
||||
)
|
||||
]
|
||||
if len(content) != len(message.content):
|
||||
message.content = content
|
||||
|
||||
|
||||
class SanitizeThinkingBlocksMiddleware(AgentMiddleware):
|
||||
"""Drop empty Anthropic thinking blocks before provider validation."""
|
||||
|
||||
def wrap_model_call(
|
||||
self,
|
||||
request: ModelRequest,
|
||||
handler: Callable[[ModelRequest], ModelResponse],
|
||||
) -> ModelCallResult:
|
||||
if _is_chat_anthropic(request.model):
|
||||
_sanitize_messages(request.messages)
|
||||
return handler(request)
|
||||
|
||||
async def awrap_model_call(
|
||||
self,
|
||||
request: ModelRequest,
|
||||
handler: Callable[[ModelRequest], Awaitable[ModelResponse]],
|
||||
) -> Any:
|
||||
if _is_chat_anthropic(request.model):
|
||||
_sanitize_messages(request.messages)
|
||||
return await handler(request)
|
||||
|
|
@ -35,6 +35,7 @@ from deepagents import create_deep_agent
|
|||
from langchain.agents.middleware import ModelCallLimitMiddleware
|
||||
|
||||
from .middleware import (
|
||||
SanitizeThinkingBlocksMiddleware,
|
||||
SanitizeToolInputsMiddleware,
|
||||
SlackAssistantStatusMiddleware,
|
||||
ToolErrorMiddleware,
|
||||
|
|
@ -805,5 +806,6 @@ async def get_reviewer_agent(config: RunnableConfig) -> Pregel:
|
|||
ToolErrorMiddleware(),
|
||||
check_message_queue_before_model,
|
||||
SlackAssistantStatusMiddleware(),
|
||||
SanitizeThinkingBlocksMiddleware(),
|
||||
],
|
||||
).with_config(config)
|
||||
|
|
|
|||
|
|
@ -47,6 +47,7 @@ from .integrations.langsmith import _configure_github_proxy
|
|||
from .middleware import (
|
||||
ModelFallbackMiddleware,
|
||||
SandboxCircuitBreakerMiddleware,
|
||||
SanitizeThinkingBlocksMiddleware,
|
||||
SanitizeToolInputsMiddleware,
|
||||
SlackAssistantStatusMiddleware,
|
||||
ToolErrorMiddleware,
|
||||
|
|
@ -518,5 +519,6 @@ async def get_agent(config: RunnableConfig) -> Pregel:
|
|||
notify_step_limit_reached,
|
||||
SandboxCircuitBreakerMiddleware(),
|
||||
*fallback_middleware,
|
||||
SanitizeThinkingBlocksMiddleware(),
|
||||
],
|
||||
).with_config(config)
|
||||
|
|
|
|||
76
tests/test_sanitize_thinking_blocks.py
Normal file
76
tests/test_sanitize_thinking_blocks.py
Normal file
|
|
@ -0,0 +1,76 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import pytest
|
||||
from langchain_anthropic import ChatAnthropic
|
||||
from langchain_core.messages import AIMessage, HumanMessage
|
||||
|
||||
from agent.middleware.sanitize_thinking_blocks import SanitizeThinkingBlocksMiddleware
|
||||
|
||||
|
||||
def _make_request(messages: list[object], model: object | None = None) -> MagicMock:
|
||||
request = MagicMock()
|
||||
request.model = model or MagicMock(spec=ChatAnthropic)
|
||||
request.messages = messages
|
||||
return request
|
||||
|
||||
|
||||
class TestSanitizeThinkingBlocksMiddleware:
|
||||
def test_drops_empty_thinking_block_for_anthropic(self) -> None:
|
||||
message = AIMessage(
|
||||
content=[
|
||||
{"type": "thinking", "signature": "abc", "thinking": ""},
|
||||
{"type": "text", "text": "ok"},
|
||||
]
|
||||
)
|
||||
request = _make_request([message])
|
||||
response = MagicMock()
|
||||
|
||||
def handler(req: object) -> object:
|
||||
assert req is request
|
||||
return response
|
||||
|
||||
result = SanitizeThinkingBlocksMiddleware().wrap_model_call(request, handler)
|
||||
|
||||
assert result is response
|
||||
assert message.content == [{"type": "text", "text": "ok"}]
|
||||
|
||||
def test_preserves_non_empty_thinking_block_for_anthropic(self) -> None:
|
||||
thinking_block = {"type": "thinking", "signature": "abc", "thinking": "reasoning"}
|
||||
text_block = {"type": "text", "text": "ok"}
|
||||
message = AIMessage(content=[thinking_block, text_block])
|
||||
request = _make_request([message])
|
||||
|
||||
SanitizeThinkingBlocksMiddleware().wrap_model_call(request, lambda req: MagicMock())
|
||||
|
||||
assert message.content == [thinking_block, text_block]
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_drops_missing_thinking_block_for_anthropic(self) -> None:
|
||||
message = AIMessage(
|
||||
content=[
|
||||
{"type": "thinking", "signature": "abc"},
|
||||
{"type": "text", "text": "ok"},
|
||||
]
|
||||
)
|
||||
request = _make_request([HumanMessage(content="hi"), message])
|
||||
response = MagicMock()
|
||||
|
||||
async def handler(req: object) -> object:
|
||||
assert req is request
|
||||
return response
|
||||
|
||||
result = await SanitizeThinkingBlocksMiddleware().awrap_model_call(request, handler)
|
||||
|
||||
assert result is response
|
||||
assert message.content == [{"type": "text", "text": "ok"}]
|
||||
|
||||
def test_ignores_non_anthropic_models(self) -> None:
|
||||
thinking_block = {"type": "thinking", "signature": "abc", "thinking": ""}
|
||||
message = AIMessage(content=[thinking_block, {"type": "text", "text": "ok"}])
|
||||
request = _make_request([message], model=MagicMock())
|
||||
|
||||
SanitizeThinkingBlocksMiddleware().wrap_model_call(request, lambda req: MagicMock())
|
||||
|
||||
assert message.content == [thinking_block, {"type": "text", "text": "ok"}]
|
||||
Loading…
Add table
Reference in a new issue