mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 22:03:14 +00:00
* wip(rebuild): core reliability spine
- remove PR-babysitting (ci_autofix + ci_monitor graph + webhook wiring)
- dispatch core: agent/dispatch.py with multitask_strategy=interrupt +
durability=sync + completion webhook; reroute all webhook + plan triggers;
drop the racy in-process lock + is_thread_active busy-check
- completion webhook: agent/completion.py + /webhooks/run-complete loopback
route for failure/timeout replies (idempotent)
Co-authored-by: open-swe[bot]
* feat(rebuild): async tools, reconcile, shared http timeouts, assembly tuning
Parallel batch on top of the reliability spine:
- async-ify all 24 tools (drop asyncio.run; requests->httpx); re-implement the
http_request/fetch_url SSRF + DNS-rebinding defense httpx-natively and harden
the IP check to 'not is_global' (+ IPv4-mapped unwrap)
- reconcile.py: stale pending-run sweep (threads.search -> per-thread runs.list
-> cancel_many), wired into the scheduler graph via task='reconcile'
- shared DEFAULT_HTTP_TIMEOUT (agent/utils/http.py) on every bare
httpx.AsyncClient() across utils/dashboard/webapp/middleware
- run budget: MODEL_CALL_RECURSION_LIMIT 5000->250
- fix stale OpenAI->Anthropic fallback id (claude-opus-4-5 -> 4-8)
- drop redundant custom repair middleware (deepagents auto-adds PatchToolCalls)
- confirm tool-result eviction + summarization auto-wired via backend
- slim system prompt ~8% (full harness-profile rewrite deferred)
Co-authored-by: open-swe[bot]
* feat(rebuild): harness-profile prompt + split webhooks out of webapp
- prompt.py: own the system prompt via a registered harness profile
(OPEN_SWE_SHARED_BASE, kept neutral so the read-only reviewer/analyzer that
share it stay safe), registered across all 4 providers; per-thread values
stay in construct_system_prompt. Assembled main-agent prompt ~6.8k -> ~3.1k
tokens (~55% smaller); de-duped PR/commit/suite/force-push guidance; dropped
ALL-CAPS markers.
- webapp.py 3325 -> 1890 LOC: moved 14 per-source handlers into
agent/webhooks/{linear,slack,github}.py; webapp re-exports them for the
routes + tests; moved handlers reach shared helpers via the webapp namespace
to preserve the test suite's monkeypatch targets.
Full suite: 1168 passing, lint clean.
Co-authored-by: open-swe[bot]
* Restore MODEL_CALL_RECURSION_LIMIT to 5000 for long-running tasks
Reverts the 250 cap from the run-budget change — long-running tasks legitimately
need many model calls. The notify_step_limit_reached safety net still fires if a
run does hit the cap, so runs end with a signal either way.
Co-authored-by: open-swe[bot]
* fix: address PR review (auth, SSRF, interrupted status, redirect headers)
- completion.py: drop `interrupted` from failure statuses — with
multitask_strategy=interrupt a follow-up ends the prior run as interrupted,
which is healthy, not a failure to report. [open-swe]
- /webhooks/run-complete: shared-secret auth — dispatch appends ?token= when
RUN_COMPLETE_WEBHOOK_SECRET is set; route verifies via hmac.compare_digest.
[corridor-security]
- SSRF: extract the URL validator to agent/utils/url_safety.py and apply it
before server-side image fetches in multimodal.fetch_image_block.
[corridor-security]
- http_request: preserve caller headers/extensions across redirect hops instead
of dropping them on the first hop. [open-swe]
Co-authored-by: open-swe[bot]
* chore: remove REBUILD_PLAN.md (planning doc, not needed in the repo)
Co-authored-by: open-swe[bot]
* fix: fail closed on run-complete webhook auth when secret unset
Corridor follow-up: verify_run_complete_token returns False (not True) when
RUN_COMPLETE_WEBHOOK_SECRET is unset, so the public route is never
unauthenticated. Logs a startup warning when the secret is absent, and dispatch
skips registering the webhook when there's no secret (no rejected callbacks).
Co-authored-by: open-swe[bot]
---------
Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
216 lines
8.4 KiB
Python
216 lines
8.4 KiB
Python
import os
|
|
from typing import Literal, TypedDict, Unpack
|
|
|
|
from langchain.chat_models import init_chat_model
|
|
|
|
from ..dashboard.options import DEFAULT_MODEL_ID
|
|
|
|
OPENAI_RESPONSES_WS_BASE_URL = "wss://api.openai.com/v1"
|
|
|
|
# Anthropic SDK default is 2; a 529 burst can outlive that. Bump to give the
|
|
# primary provider a fair chance before the fallback middleware kicks in.
|
|
DEFAULT_MAX_RETRIES = 6
|
|
|
|
OpenAIReasoningEffort = Literal["none", "low", "medium", "high", "xhigh"]
|
|
# OpenAI's Responses API only returns human-readable reasoning text when a
|
|
# summary is requested; without it, reasoning happens silently (billed in
|
|
# output tokens) and the reasoning content block arrives empty.
|
|
OpenAIReasoningSummary = Literal["auto", "concise", "detailed"]
|
|
AnthropicThinkingType = Literal["adaptive"]
|
|
AnthropicThinkingDisplay = Literal["summarized", "omitted"]
|
|
AnthropicEffort = Literal["low", "medium", "high", "xhigh", "max"]
|
|
GoogleThinkingLevel = Literal["minimal", "low", "medium", "high"]
|
|
FireworksReasoningEffort = Literal["none", "low", "medium", "high", "xhigh", "max"]
|
|
|
|
|
|
class OpenAIReasoning(TypedDict, total=False):
|
|
effort: OpenAIReasoningEffort
|
|
summary: OpenAIReasoningSummary
|
|
|
|
|
|
DEFAULT_LLM_REASONING: "OpenAIReasoning" = {"effort": "medium", "summary": "auto"}
|
|
|
|
|
|
class AnthropicThinking(TypedDict, total=False):
|
|
type: AnthropicThinkingType
|
|
display: AnthropicThinkingDisplay
|
|
|
|
|
|
class ModelKwargs(TypedDict, total=False):
|
|
max_tokens: int | None
|
|
reasoning: OpenAIReasoning | None
|
|
thinking: AnthropicThinking | None
|
|
effort: AnthropicEffort | None
|
|
thinking_level: GoogleThinkingLevel | None
|
|
temperature: float | None
|
|
max_retries: int | None
|
|
model_kwargs: dict[str, object] | None
|
|
|
|
|
|
_ANTHROPIC_EFFORTS: set[AnthropicEffort] = {"low", "medium", "high", "xhigh", "max"}
|
|
|
|
|
|
def make_model(model_id: str, **kwargs: Unpack[ModelKwargs]):
|
|
model_kwargs: dict[str, object] = kwargs.copy()
|
|
model_kwargs.setdefault("max_retries", DEFAULT_MAX_RETRIES)
|
|
|
|
if model_id.startswith("openai:"):
|
|
model_kwargs["base_url"] = OPENAI_RESPONSES_WS_BASE_URL
|
|
model_kwargs["use_responses_api"] = True
|
|
|
|
return init_chat_model(model=model_id, **model_kwargs)
|
|
|
|
|
|
def fallback_model_id_for(primary_model_id: str) -> str | None:
|
|
"""Return the cross-provider fallback model id for a given primary, if any.
|
|
|
|
Anthropic primaries fall back to OpenAI and vice versa. Returns ``None``
|
|
when the provider has no configured cross-provider fallback (e.g. Google,
|
|
local, or self-hosted providers we don't want to silently route off-host).
|
|
"""
|
|
if primary_model_id.startswith("anthropic:"):
|
|
return "openai:gpt-5.5"
|
|
if primary_model_id.startswith("openai:"):
|
|
return "anthropic:claude-opus-4-8"
|
|
return None
|
|
|
|
|
|
def is_gemini_3_family(model_id: str) -> bool:
|
|
model_name = model_id.split(":", 1)[-1]
|
|
return model_name.startswith("gemini-3")
|
|
|
|
|
|
def openai_reasoning_for(
|
|
profile_effort: str | None,
|
|
*,
|
|
default_effort: OpenAIReasoningEffort | None = None,
|
|
) -> OpenAIReasoning | None:
|
|
"""Return an OpenAI reasoning kwarg from a profile effort string.
|
|
|
|
Requests ``summary: "auto"`` for every reasoning effort so the Responses
|
|
API emits visible reasoning text. ``effort: "none"`` disables reasoning
|
|
entirely, so no summary is attached.
|
|
"""
|
|
effort = profile_effort or default_effort or DEFAULT_LLM_REASONING.get("effort")
|
|
if effort == "none":
|
|
return {"effort": "none"}
|
|
if effort == "low":
|
|
return {"effort": "low", "summary": "auto"}
|
|
if effort == "medium":
|
|
return {"effort": "medium", "summary": "auto"}
|
|
if effort == "high":
|
|
return {"effort": "high", "summary": "auto"}
|
|
if effort == "xhigh":
|
|
return {"effort": "xhigh", "summary": "auto"}
|
|
return None
|
|
|
|
|
|
def anthropic_thinking_for(profile_effort: str | None) -> AnthropicThinking | None:
|
|
if profile_effort in _ANTHROPIC_EFFORTS:
|
|
# `display: "summarized"` makes Opus 4.7+ return the (summarized) reasoning
|
|
# text in the response. The adaptive default is "omitted", which streams a
|
|
# reasoning block carrying only a signature and no visible thinking — so the
|
|
# dashboard never has any text to render.
|
|
return {"type": "adaptive", "display": "summarized"}
|
|
return None
|
|
|
|
|
|
def anthropic_effort_for(profile_effort: str | None) -> AnthropicEffort | None:
|
|
if profile_effort in _ANTHROPIC_EFFORTS:
|
|
return profile_effort
|
|
return None
|
|
|
|
|
|
def fireworks_reasoning_effort_for(profile_effort: str | None) -> FireworksReasoningEffort | None:
|
|
"""Map profile effort to a Fireworks ``reasoning_effort`` value.
|
|
|
|
Fireworks' OpenAI-compatible API accepts ``reasoning_effort`` on its reasoning
|
|
models. ``none`` disables reasoning; ``xhigh``/``max`` are only honored by models
|
|
that advertise them (e.g. DeepSeek V4 Pro). The per-model ``efforts`` lists in
|
|
``dashboard/options.py`` gate which values can actually reach this function.
|
|
"""
|
|
if profile_effort == "none":
|
|
return "none"
|
|
if profile_effort == "low":
|
|
return "low"
|
|
if profile_effort == "medium":
|
|
return "medium"
|
|
if profile_effort == "high":
|
|
return "high"
|
|
if profile_effort == "xhigh":
|
|
return "xhigh"
|
|
if profile_effort == "max":
|
|
return "max"
|
|
return None
|
|
|
|
|
|
def google_thinking_level_for(profile_effort: str | None) -> GoogleThinkingLevel | None:
|
|
"""Map profile effort to Gemini 3+ ``thinking_level``."""
|
|
if profile_effort in ("minimal", "none"):
|
|
return "minimal"
|
|
if profile_effort == "low":
|
|
return "low"
|
|
if profile_effort == "medium":
|
|
return "medium"
|
|
if profile_effort in ("high", "xhigh", "max"):
|
|
return "high"
|
|
return None
|
|
|
|
|
|
def provider_model_kwargs(
|
|
model_id: str,
|
|
profile_effort: str | None,
|
|
*,
|
|
max_tokens: int,
|
|
openai_reasoning_default: OpenAIReasoning | None = None,
|
|
) -> ModelKwargs:
|
|
"""Build provider-specific kwargs for ``make_model`` from a model id and effort."""
|
|
kwargs: ModelKwargs = {"max_tokens": max_tokens}
|
|
if model_id.startswith("openai:"):
|
|
reasoning = openai_reasoning_for(profile_effort)
|
|
if reasoning is not None:
|
|
kwargs["reasoning"] = reasoning
|
|
elif openai_reasoning_default is not None:
|
|
kwargs["reasoning"] = openai_reasoning_default
|
|
elif model_id.startswith("anthropic:"):
|
|
thinking = anthropic_thinking_for(profile_effort)
|
|
if thinking is not None:
|
|
kwargs["thinking"] = thinking
|
|
effort = anthropic_effort_for(profile_effort)
|
|
if effort is not None:
|
|
kwargs["effort"] = effort
|
|
elif model_id.startswith("google_genai:") and is_gemini_3_family(model_id):
|
|
thinking_level = google_thinking_level_for(profile_effort)
|
|
if thinking_level is not None:
|
|
kwargs["thinking_level"] = thinking_level
|
|
elif model_id.startswith("fireworks:"):
|
|
effort = fireworks_reasoning_effort_for(profile_effort)
|
|
if effort is not None:
|
|
kwargs["model_kwargs"] = {"reasoning_effort": effort}
|
|
return kwargs
|
|
|
|
|
|
def validate_local_dev_llm_config() -> None:
|
|
"""Validate API keys for the locally configured default model.
|
|
|
|
This check only runs in localhost development environments and is
|
|
intended to catch missing credentials for the default model specified
|
|
via LLM_MODEL_ID/DEFAULT_MODEL_ID. Runtime model selection may come
|
|
from team, profile, or thread configuration and is not validated here.
|
|
"""
|
|
dashboard_url = os.environ.get("DASHBOARD_BASE_URL", "")
|
|
if not dashboard_url.startswith("http://localhost"):
|
|
return
|
|
|
|
model_id = os.environ.get("LLM_MODEL_ID", DEFAULT_MODEL_ID)
|
|
|
|
if model_id.startswith("openai:") and not os.environ.get("OPENAI_API_KEY"):
|
|
raise ValueError(f"OPENAI_API_KEY is required for configured model {model_id}")
|
|
elif model_id.startswith("anthropic:") and not os.environ.get("ANTHROPIC_API_KEY"):
|
|
raise ValueError(f"ANTHROPIC_API_KEY is required for configured model {model_id}")
|
|
elif model_id.startswith("google_genai:") and not os.environ.get("GOOGLE_API_KEY"):
|
|
raise ValueError(f"GOOGLE_API_KEY is required for configured model {model_id}")
|
|
elif model_id.startswith("groq:") and not os.environ.get("GROQ_API_KEY"):
|
|
raise ValueError(f"GROQ_API_KEY is required for configured model {model_id}")
|
|
elif model_id.startswith("fireworks:") and not os.environ.get("FIREWORKS_API_KEY"):
|
|
raise ValueError(f"FIREWORKS_API_KEY is required for configured model {model_id}")
|