mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 18:33:15 +00:00
* wip(rebuild): core reliability spine
- remove PR-babysitting (ci_autofix + ci_monitor graph + webhook wiring)
- dispatch core: agent/dispatch.py with multitask_strategy=interrupt +
durability=sync + completion webhook; reroute all webhook + plan triggers;
drop the racy in-process lock + is_thread_active busy-check
- completion webhook: agent/completion.py + /webhooks/run-complete loopback
route for failure/timeout replies (idempotent)
Co-authored-by: open-swe[bot]
* feat(rebuild): async tools, reconcile, shared http timeouts, assembly tuning
Parallel batch on top of the reliability spine:
- async-ify all 24 tools (drop asyncio.run; requests->httpx); re-implement the
http_request/fetch_url SSRF + DNS-rebinding defense httpx-natively and harden
the IP check to 'not is_global' (+ IPv4-mapped unwrap)
- reconcile.py: stale pending-run sweep (threads.search -> per-thread runs.list
-> cancel_many), wired into the scheduler graph via task='reconcile'
- shared DEFAULT_HTTP_TIMEOUT (agent/utils/http.py) on every bare
httpx.AsyncClient() across utils/dashboard/webapp/middleware
- run budget: MODEL_CALL_RECURSION_LIMIT 5000->250
- fix stale OpenAI->Anthropic fallback id (claude-opus-4-5 -> 4-8)
- drop redundant custom repair middleware (deepagents auto-adds PatchToolCalls)
- confirm tool-result eviction + summarization auto-wired via backend
- slim system prompt ~8% (full harness-profile rewrite deferred)
Co-authored-by: open-swe[bot]
* feat(rebuild): harness-profile prompt + split webhooks out of webapp
- prompt.py: own the system prompt via a registered harness profile
(OPEN_SWE_SHARED_BASE, kept neutral so the read-only reviewer/analyzer that
share it stay safe), registered across all 4 providers; per-thread values
stay in construct_system_prompt. Assembled main-agent prompt ~6.8k -> ~3.1k
tokens (~55% smaller); de-duped PR/commit/suite/force-push guidance; dropped
ALL-CAPS markers.
- webapp.py 3325 -> 1890 LOC: moved 14 per-source handlers into
agent/webhooks/{linear,slack,github}.py; webapp re-exports them for the
routes + tests; moved handlers reach shared helpers via the webapp namespace
to preserve the test suite's monkeypatch targets.
Full suite: 1168 passing, lint clean.
Co-authored-by: open-swe[bot]
* Restore MODEL_CALL_RECURSION_LIMIT to 5000 for long-running tasks
Reverts the 250 cap from the run-budget change — long-running tasks legitimately
need many model calls. The notify_step_limit_reached safety net still fires if a
run does hit the cap, so runs end with a signal either way.
Co-authored-by: open-swe[bot]
* fix: address PR review (auth, SSRF, interrupted status, redirect headers)
- completion.py: drop `interrupted` from failure statuses — with
multitask_strategy=interrupt a follow-up ends the prior run as interrupted,
which is healthy, not a failure to report. [open-swe]
- /webhooks/run-complete: shared-secret auth — dispatch appends ?token= when
RUN_COMPLETE_WEBHOOK_SECRET is set; route verifies via hmac.compare_digest.
[corridor-security]
- SSRF: extract the URL validator to agent/utils/url_safety.py and apply it
before server-side image fetches in multimodal.fetch_image_block.
[corridor-security]
- http_request: preserve caller headers/extensions across redirect hops instead
of dropping them on the first hop. [open-swe]
Co-authored-by: open-swe[bot]
* chore: remove REBUILD_PLAN.md (planning doc, not needed in the repo)
Co-authored-by: open-swe[bot]
* fix: fail closed on run-complete webhook auth when secret unset
Corridor follow-up: verify_run_complete_token returns False (not True) when
RUN_COMPLETE_WEBHOOK_SECRET is unset, so the public route is never
unauthenticated. Logs a startup warning when the secret is absent, and dispatch
skips registering the webhook when there's no secret (no rejected callbacks).
Co-authored-by: open-swe[bot]
---------
Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
188 lines
7.6 KiB
Python
188 lines
7.6 KiB
Python
"""Tool: ``add_finding``. Records one review finding on the reviewer thread."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import Any
|
|
|
|
from langgraph.config import get_config
|
|
|
|
from ..reviewer_diff import is_range_in_diff
|
|
from ..reviewer_findings import (
|
|
DEFAULT_FINDING_TITLE,
|
|
MAX_SUGGESTION_LINES,
|
|
Confidence,
|
|
DiffSide,
|
|
Finding,
|
|
ReviewerThreadMissingError,
|
|
Severity,
|
|
append_finding,
|
|
clip_suggestion,
|
|
get_thread_id_from_runtime,
|
|
new_finding,
|
|
normalize_finding_title,
|
|
resolve_review_head_sha,
|
|
thread_missing_tool_result,
|
|
)
|
|
|
|
|
|
async def add_finding(
|
|
severity: str,
|
|
confidence: str,
|
|
category: str,
|
|
file: str,
|
|
title: str,
|
|
description: str,
|
|
start_line: int | None = None,
|
|
end_line: int | None = None,
|
|
suggestion: str | None = None,
|
|
side: str = "RIGHT",
|
|
) -> dict[str, Any]:
|
|
"""Record a review finding on the reviewer thread.
|
|
|
|
Findings persist on the reviewer thread's metadata so they survive sandbox
|
|
eviction and are queryable across runs by the watch-mode reconciliation
|
|
flow and the future UI.
|
|
|
|
**When to use:** Once per distinct issue you find while reviewing the
|
|
diff. Prefer one finding per issue, with a concise generated ``title`` that
|
|
names the failure mode, a clear ``description`` body, and, when you can
|
|
offer a concrete fix, a ``suggestion`` that exactly replaces lines
|
|
``start_line..end_line``.
|
|
|
|
**In-diff only:** ``start_line..end_line`` must be inside the PR diff.
|
|
Findings anchored to lines outside the diff are rejected (out-of-diff
|
|
findings are disabled). File-level findings (both ``start_line`` and
|
|
``end_line`` None) are accepted but won't render as inline GitHub
|
|
comments — only use when the issue truly isn't anchored to a line.
|
|
|
|
Args:
|
|
severity: One of ``low``, ``medium``, ``high``, ``critical``.
|
|
confidence: One of ``low``, ``medium``, ``high``.
|
|
category: Short category label (``correctness``, ``security``, ``perf``,
|
|
``style``, ``flag``, etc.). Free-form; used for grouping in the UI.
|
|
file: Repo-relative path of the file the finding refers to.
|
|
title: Concise generated headline for the finding. Name the failure mode
|
|
in roughly 4-10 words; do not copy or truncate the description.
|
|
description: Markdown body the user sees. Do not repeat ``title`` as the
|
|
first line.
|
|
start_line: 1-based line in the new (post-PR) file where the
|
|
relevant range begins. For a single-line finding, this is the
|
|
line the issue is about. For a multi-line finding, this is the
|
|
first line of the relevant range. Omit (with ``end_line``) for
|
|
file-level findings.
|
|
end_line: 1-based line where the relevant range ends. GitHub
|
|
anchors the inline comment at ``end_line`` and renders the
|
|
``start_line..end_line`` span as the highlighted snippet, so
|
|
choose ``end_line`` as the *last* line that matters — typically
|
|
the line the comment is most directly about. For a single-line
|
|
finding, set ``end_line == start_line`` (or omit it). Prefer
|
|
the natural range of the issue over a single line: GitHub
|
|
shows context above ``end_line``, so a one-line anchor often
|
|
buries the issue under unrelated context. Defaults to
|
|
``start_line`` when omitted.
|
|
suggestion: Replacement text for ``start_line..end_line``. When set,
|
|
the published GitHub comment includes a ```suggestion``` block so
|
|
the user can click "Commit suggestion". **Only set this for small,
|
|
obvious fixes that fit in 4 lines or fewer** (e.g. a one-liner
|
|
rename, a missing guard, a typo). Longer suggestions are dropped
|
|
because they read as rewrites rather than reviews — leave those
|
|
cases as a description-only finding so the author can decide how
|
|
to fix it.
|
|
side: ``RIGHT`` (post-PR file, default) or ``LEFT`` (base file). Almost
|
|
always ``RIGHT``.
|
|
|
|
Returns:
|
|
Dictionary with ``success``, ``finding_id`` and (on rejection) ``error``.
|
|
"""
|
|
if start_line is not None and end_line is None:
|
|
end_line = start_line
|
|
if start_line is None and end_line is not None:
|
|
start_line = end_line
|
|
|
|
normalized_title = normalize_finding_title(title)
|
|
if normalized_title == DEFAULT_FINDING_TITLE:
|
|
return {"success": False, "error": "title must be a non-empty generated headline"}
|
|
|
|
if severity not in {"low", "medium", "high", "critical"}:
|
|
return {"success": False, "error": f"Invalid severity: {severity}"}
|
|
if confidence not in {"low", "medium", "high"}:
|
|
return {"success": False, "error": f"Invalid confidence: {confidence}"}
|
|
if side not in {"LEFT", "RIGHT"}:
|
|
return {"success": False, "error": f"Invalid side: {side}"}
|
|
if start_line is not None and end_line is not None and end_line < start_line:
|
|
return {"success": False, "error": "end_line must be >= start_line"}
|
|
|
|
config = get_config()
|
|
configurable = config.get("configurable", {}) if isinstance(config, dict) else {}
|
|
diff_line_set = configurable.get("diff_line_set") if isinstance(configurable, dict) else None
|
|
diff_text = configurable.get("diff_text", "") if isinstance(configurable, dict) else ""
|
|
|
|
in_diff = not isinstance(diff_line_set, dict) or is_range_in_diff(
|
|
diff_line_set, file, start_line, end_line, side=_cast_side(side)
|
|
)
|
|
if not in_diff:
|
|
return {
|
|
"success": False,
|
|
"in_diff": False,
|
|
"error": (
|
|
"Out-of-diff findings are disabled. This finding's lines are not "
|
|
"part of this PR's diff. Only file findings anchored to a line the "
|
|
"PR changed. Do not re-anchor or retry."
|
|
),
|
|
}
|
|
|
|
diff_hunk: str | None = None
|
|
if isinstance(diff_text, str) and diff_text:
|
|
from ..reviewer_diff import extract_diff_hunk
|
|
|
|
diff_hunk = extract_diff_hunk(diff_text, file, start_line, end_line)
|
|
|
|
clipped_suggestion, suggestion_dropped = clip_suggestion(suggestion)
|
|
|
|
thread_id = get_thread_id_from_runtime()
|
|
try:
|
|
head_sha = await resolve_review_head_sha(thread_id, configurable)
|
|
except ReviewerThreadMissingError as exc:
|
|
return thread_missing_tool_result(exc)
|
|
|
|
finding: Finding = new_finding(
|
|
severity=_cast_severity(severity),
|
|
confidence=_cast_confidence(confidence),
|
|
category=category,
|
|
file=file,
|
|
start_line=start_line,
|
|
end_line=end_line,
|
|
description=description,
|
|
sha=head_sha,
|
|
title=normalized_title,
|
|
side=_cast_side(side),
|
|
suggestion=clipped_suggestion,
|
|
diff_hunk=diff_hunk,
|
|
in_diff=in_diff,
|
|
)
|
|
|
|
try:
|
|
await append_finding(thread_id, finding)
|
|
except ReviewerThreadMissingError as exc:
|
|
return thread_missing_tool_result(exc)
|
|
result: dict[str, Any] = {"success": True, "finding_id": finding["id"]}
|
|
if suggestion_dropped:
|
|
result["suggestion_dropped"] = True
|
|
result["warning"] = (
|
|
f"Suggestion exceeded the {MAX_SUGGESTION_LINES}-line cap and was "
|
|
"dropped — the finding was recorded with description only. Only "
|
|
"include `suggestion` for small, obvious fixes."
|
|
)
|
|
return result
|
|
|
|
|
|
def _cast_severity(value: str) -> Severity:
|
|
return value # type: ignore[return-value]
|
|
|
|
|
|
def _cast_confidence(value: str) -> Confidence:
|
|
return value # type: ignore[return-value]
|
|
|
|
|
|
def _cast_side(value: str) -> DiffSide:
|
|
return value # type: ignore[return-value]
|