mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-10-05 21:12:13 +00:00
fix: sanitize malformed Anthropic thinking blocks (#1357)
Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> Co-authored-by: Johannes du Plessis <51395795+johannes117@users.noreply.github.com>
This commit is contained in:
parent
eb92947806
commit
65acc4c9d3
6 changed files with 152 additions and 2 deletions
|
|
@ -68,9 +68,10 @@ Configured in `agent/server.py:get_agent`, runs around every model call (in this
|
||||||
6. `ensure_no_empty_msg` — guards against empty assistant messages that some providers reject.
|
6. `ensure_no_empty_msg` — guards against empty assistant messages that some providers reject.
|
||||||
7. `notify_step_limit_reached` — after-agent hook that posts a Slack reply when the agent hits the step limit, so the user gets a clear signal instead of silence.
|
7. `notify_step_limit_reached` — after-agent hook that posts a Slack reply when the agent hits the step limit, so the user gets a clear signal instead of silence.
|
||||||
8. `SandboxCircuitBreakerMiddleware` — trips the agent out of repeated sandbox failures instead of looping.
|
8. `SandboxCircuitBreakerMiddleware` — trips the agent out of repeated sandbox failures instead of looping.
|
||||||
9. `ModelFallbackMiddleware` (optional, last) — added only when `LLM_FALLBACK_MODEL_ID` or the per-model default fallback differs from the primary model.
|
9. `ModelFallbackMiddleware` (optional) — added only when `LLM_FALLBACK_MODEL_ID` or the per-model default fallback differs from the primary model.
|
||||||
|
10. `SanitizeThinkingBlocksMiddleware` — strips malformed empty Anthropic thinking blocks immediately before provider calls.
|
||||||
|
|
||||||
Other middleware exists in `agent/middleware/` (`ExcludeToolsMiddleware`) but isn't wired into the default agent. The reviewer uses a leaner stack: `SanitizeToolInputsMiddleware`, `ModelCallLimitMiddleware`, `ToolErrorMiddleware`, `SlackAssistantStatusMiddleware`.
|
Other middleware exists in `agent/middleware/` (`ExcludeToolsMiddleware`) but isn't wired into the default agent. The reviewer uses a leaner stack: `SanitizeToolInputsMiddleware`, `ModelCallLimitMiddleware`, `ToolErrorMiddleware`, `SlackAssistantStatusMiddleware`, `SanitizeThinkingBlocksMiddleware`.
|
||||||
|
|
||||||
There is intentionally no after-agent safety net that opens a PR for the agent. The agent itself is responsible for committing, pushing, opening/updating the draft PR, and replying in the source channel — all via `GH_TOKEN=dummy gh` and `slack_thread_reply` / `linear_comment`.
|
There is intentionally no after-agent safety net that opens a PR for the agent. The agent itself is responsible for committing, pushing, opening/updating the draft PR, and replying in the source channel — all via `GH_TOKEN=dummy gh` and `slack_thread_reply` / `linear_comment`.
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -5,12 +5,14 @@ from .model_fallback import ModelFallbackMiddleware
|
||||||
from .notify_step_limit import notify_step_limit_reached
|
from .notify_step_limit import notify_step_limit_reached
|
||||||
from .refresh_slack_status import SlackAssistantStatusMiddleware
|
from .refresh_slack_status import SlackAssistantStatusMiddleware
|
||||||
from .sandbox_circuit_breaker import SandboxCircuitBreakerMiddleware
|
from .sandbox_circuit_breaker import SandboxCircuitBreakerMiddleware
|
||||||
|
from .sanitize_thinking_blocks import SanitizeThinkingBlocksMiddleware
|
||||||
from .sanitize_tool_inputs import SanitizeToolInputsMiddleware
|
from .sanitize_tool_inputs import SanitizeToolInputsMiddleware
|
||||||
from .tool_error_handler import ToolErrorMiddleware
|
from .tool_error_handler import ToolErrorMiddleware
|
||||||
|
|
||||||
__all__ = [
|
__all__ = [
|
||||||
"ExcludeToolsMiddleware",
|
"ExcludeToolsMiddleware",
|
||||||
"ModelFallbackMiddleware",
|
"ModelFallbackMiddleware",
|
||||||
|
"SanitizeThinkingBlocksMiddleware",
|
||||||
"SanitizeToolInputsMiddleware",
|
"SanitizeToolInputsMiddleware",
|
||||||
"ToolErrorMiddleware",
|
"ToolErrorMiddleware",
|
||||||
"SandboxCircuitBreakerMiddleware",
|
"SandboxCircuitBreakerMiddleware",
|
||||||
|
|
|
||||||
67
agent/middleware/sanitize_thinking_blocks.py
Normal file
67
agent/middleware/sanitize_thinking_blocks.py
Normal file
|
|
@ -0,0 +1,67 @@
|
||||||
|
"""Middleware that removes malformed Anthropic thinking blocks before model calls."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from collections.abc import Awaitable, Callable
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from langchain.agents.middleware import AgentMiddleware
|
||||||
|
from langchain.agents.middleware.types import ModelCallResult, ModelRequest, ModelResponse
|
||||||
|
from langchain_anthropic import ChatAnthropic
|
||||||
|
from langchain_core.messages import AIMessage
|
||||||
|
|
||||||
|
|
||||||
|
def _is_chat_anthropic(model: object) -> bool:
|
||||||
|
seen: set[int] = set()
|
||||||
|
current = model
|
||||||
|
for _ in range(10):
|
||||||
|
if isinstance(current, ChatAnthropic):
|
||||||
|
return True
|
||||||
|
current_id = id(current)
|
||||||
|
if current_id in seen:
|
||||||
|
return False
|
||||||
|
seen.add(current_id)
|
||||||
|
bound = getattr(current, "bound", None)
|
||||||
|
if bound is None or bound is current:
|
||||||
|
return False
|
||||||
|
current = bound
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def _sanitize_messages(messages: list[Any]) -> None:
|
||||||
|
for message in messages:
|
||||||
|
if not isinstance(message, AIMessage) or not isinstance(message.content, list):
|
||||||
|
continue
|
||||||
|
content = [
|
||||||
|
block
|
||||||
|
for block in message.content
|
||||||
|
if not (
|
||||||
|
isinstance(block, dict)
|
||||||
|
and block.get("type") == "thinking"
|
||||||
|
and not block.get("thinking")
|
||||||
|
)
|
||||||
|
]
|
||||||
|
if len(content) != len(message.content):
|
||||||
|
message.content = content
|
||||||
|
|
||||||
|
|
||||||
|
class SanitizeThinkingBlocksMiddleware(AgentMiddleware):
|
||||||
|
"""Drop empty Anthropic thinking blocks before provider validation."""
|
||||||
|
|
||||||
|
def wrap_model_call(
|
||||||
|
self,
|
||||||
|
request: ModelRequest,
|
||||||
|
handler: Callable[[ModelRequest], ModelResponse],
|
||||||
|
) -> ModelCallResult:
|
||||||
|
if _is_chat_anthropic(request.model):
|
||||||
|
_sanitize_messages(request.messages)
|
||||||
|
return handler(request)
|
||||||
|
|
||||||
|
async def awrap_model_call(
|
||||||
|
self,
|
||||||
|
request: ModelRequest,
|
||||||
|
handler: Callable[[ModelRequest], Awaitable[ModelResponse]],
|
||||||
|
) -> Any:
|
||||||
|
if _is_chat_anthropic(request.model):
|
||||||
|
_sanitize_messages(request.messages)
|
||||||
|
return await handler(request)
|
||||||
|
|
@ -35,6 +35,7 @@ from deepagents import create_deep_agent
|
||||||
from langchain.agents.middleware import ModelCallLimitMiddleware
|
from langchain.agents.middleware import ModelCallLimitMiddleware
|
||||||
|
|
||||||
from .middleware import (
|
from .middleware import (
|
||||||
|
SanitizeThinkingBlocksMiddleware,
|
||||||
SanitizeToolInputsMiddleware,
|
SanitizeToolInputsMiddleware,
|
||||||
SlackAssistantStatusMiddleware,
|
SlackAssistantStatusMiddleware,
|
||||||
ToolErrorMiddleware,
|
ToolErrorMiddleware,
|
||||||
|
|
@ -805,5 +806,6 @@ async def get_reviewer_agent(config: RunnableConfig) -> Pregel:
|
||||||
ToolErrorMiddleware(),
|
ToolErrorMiddleware(),
|
||||||
check_message_queue_before_model,
|
check_message_queue_before_model,
|
||||||
SlackAssistantStatusMiddleware(),
|
SlackAssistantStatusMiddleware(),
|
||||||
|
SanitizeThinkingBlocksMiddleware(),
|
||||||
],
|
],
|
||||||
).with_config(config)
|
).with_config(config)
|
||||||
|
|
|
||||||
|
|
@ -47,6 +47,7 @@ from .integrations.langsmith import _configure_github_proxy
|
||||||
from .middleware import (
|
from .middleware import (
|
||||||
ModelFallbackMiddleware,
|
ModelFallbackMiddleware,
|
||||||
SandboxCircuitBreakerMiddleware,
|
SandboxCircuitBreakerMiddleware,
|
||||||
|
SanitizeThinkingBlocksMiddleware,
|
||||||
SanitizeToolInputsMiddleware,
|
SanitizeToolInputsMiddleware,
|
||||||
SlackAssistantStatusMiddleware,
|
SlackAssistantStatusMiddleware,
|
||||||
ToolErrorMiddleware,
|
ToolErrorMiddleware,
|
||||||
|
|
@ -518,5 +519,6 @@ async def get_agent(config: RunnableConfig) -> Pregel:
|
||||||
notify_step_limit_reached,
|
notify_step_limit_reached,
|
||||||
SandboxCircuitBreakerMiddleware(),
|
SandboxCircuitBreakerMiddleware(),
|
||||||
*fallback_middleware,
|
*fallback_middleware,
|
||||||
|
SanitizeThinkingBlocksMiddleware(),
|
||||||
],
|
],
|
||||||
).with_config(config)
|
).with_config(config)
|
||||||
|
|
|
||||||
76
tests/test_sanitize_thinking_blocks.py
Normal file
76
tests/test_sanitize_thinking_blocks.py
Normal file
|
|
@ -0,0 +1,76 @@
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from unittest.mock import MagicMock
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from langchain_anthropic import ChatAnthropic
|
||||||
|
from langchain_core.messages import AIMessage, HumanMessage
|
||||||
|
|
||||||
|
from agent.middleware.sanitize_thinking_blocks import SanitizeThinkingBlocksMiddleware
|
||||||
|
|
||||||
|
|
||||||
|
def _make_request(messages: list[object], model: object | None = None) -> MagicMock:
|
||||||
|
request = MagicMock()
|
||||||
|
request.model = model or MagicMock(spec=ChatAnthropic)
|
||||||
|
request.messages = messages
|
||||||
|
return request
|
||||||
|
|
||||||
|
|
||||||
|
class TestSanitizeThinkingBlocksMiddleware:
|
||||||
|
def test_drops_empty_thinking_block_for_anthropic(self) -> None:
|
||||||
|
message = AIMessage(
|
||||||
|
content=[
|
||||||
|
{"type": "thinking", "signature": "abc", "thinking": ""},
|
||||||
|
{"type": "text", "text": "ok"},
|
||||||
|
]
|
||||||
|
)
|
||||||
|
request = _make_request([message])
|
||||||
|
response = MagicMock()
|
||||||
|
|
||||||
|
def handler(req: object) -> object:
|
||||||
|
assert req is request
|
||||||
|
return response
|
||||||
|
|
||||||
|
result = SanitizeThinkingBlocksMiddleware().wrap_model_call(request, handler)
|
||||||
|
|
||||||
|
assert result is response
|
||||||
|
assert message.content == [{"type": "text", "text": "ok"}]
|
||||||
|
|
||||||
|
def test_preserves_non_empty_thinking_block_for_anthropic(self) -> None:
|
||||||
|
thinking_block = {"type": "thinking", "signature": "abc", "thinking": "reasoning"}
|
||||||
|
text_block = {"type": "text", "text": "ok"}
|
||||||
|
message = AIMessage(content=[thinking_block, text_block])
|
||||||
|
request = _make_request([message])
|
||||||
|
|
||||||
|
SanitizeThinkingBlocksMiddleware().wrap_model_call(request, lambda req: MagicMock())
|
||||||
|
|
||||||
|
assert message.content == [thinking_block, text_block]
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_async_drops_missing_thinking_block_for_anthropic(self) -> None:
|
||||||
|
message = AIMessage(
|
||||||
|
content=[
|
||||||
|
{"type": "thinking", "signature": "abc"},
|
||||||
|
{"type": "text", "text": "ok"},
|
||||||
|
]
|
||||||
|
)
|
||||||
|
request = _make_request([HumanMessage(content="hi"), message])
|
||||||
|
response = MagicMock()
|
||||||
|
|
||||||
|
async def handler(req: object) -> object:
|
||||||
|
assert req is request
|
||||||
|
return response
|
||||||
|
|
||||||
|
result = await SanitizeThinkingBlocksMiddleware().awrap_model_call(request, handler)
|
||||||
|
|
||||||
|
assert result is response
|
||||||
|
assert message.content == [{"type": "text", "text": "ok"}]
|
||||||
|
|
||||||
|
def test_ignores_non_anthropic_models(self) -> None:
|
||||||
|
thinking_block = {"type": "thinking", "signature": "abc", "thinking": ""}
|
||||||
|
message = AIMessage(content=[thinking_block, {"type": "text", "text": "ok"}])
|
||||||
|
request = _make_request([message], model=MagicMock())
|
||||||
|
|
||||||
|
SanitizeThinkingBlocksMiddleware().wrap_model_call(request, lambda req: MagicMock())
|
||||||
|
|
||||||
|
assert message.content == [thinking_block, {"type": "text", "text": "ok"}]
|
||||||
Loading…
Add table
Reference in a new issue