mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 16:13:15 +00:00
Some checks are pending
CI / Lint (push) Waiting to run
CI / Format check (push) Waiting to run
CI / Typecheck (push) Waiting to run
CI / Unit tests (push) Waiting to run
CI / Playwright E2E (push) Waiting to run
CI / Docker build smoke (push) Waiting to run
CI / Triage ledger up to date (push) Waiting to run
CI / ui bun.lock in sync (push) Waiting to run
* feat: surface attributed PR creation failures Port upstream #1659: adds PullRequestCreationGuardMiddleware that blocks shell fallbacks (gh pr create, gh api /pulls, curl) when open_pull_request fails, keeping failures visible. Also adds preflight branch/repo visibility checks in open_pull_request with structured failure payloads, and updates the prompt to forbid PR creation fallbacks. Refs: #134 * fix: fall back to core GitHub App scope when optional grants missing (#1701) * fix: fall back to core GitHub App scope when optional grants missing Proxy-token minting requested workflows:write and actions:read in the permission set used for every sandbox. GitHub 422s a token request that asks for a permission the installation hasn't granted, so any install without workflows:write failed to mint a token and every run died in before-agent setup with "GitHub App installation token is unavailable". _resolve_proxy_token now walks a permission ladder (full -> +workflows -> core) and returns the first scope that mints, recording the granted scope so hourly proxy refreshes stay consistent. A missing optional grant now degrades to the install-time core scope instead of failing the run; workflow-file HITL pushes still require workflows:write and fail at push time when it is absent. * refactor: flatten proxy-token ladder loop with continue --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> (cherry picked from commit f53caff1aa24a7b29d851b267aa3bdfe62c1e935) Sea Haven fork deviation: upstream #1701 folds workflows:write into the standing BASE/RUNTIME scope. This fork deliberately keeps workflows:write OUT of the standing permission ladder (RUNTIME = core + actions:read; LADDER = (RUNTIME, CORE)) so the sandbox proxy token cannot push .github/workflows/* during normal operation. workflows:write is minted only transiently by WorkflowPushGuardMiddleware for an approved HITL push and dropped on restore, preserving token scope as a backstop for the workflow- push approval control. Security-reviewed (agentic fan-out + GPT-4.1 cross review); the standing-scope-carries-workflows:write bypass was blocked. * fix(open-swe): harden proxy-token restore and mint error handling Two low-severity follow-ups from the security review of the #1701 port. Restore the recorded baseline scope after a workflow-push elevation instead of a hardcoded RUNTIME. An install granted workflows:write but not actions:read resolves its standing token to core; hardcoding RUNTIME on restore requested the ungranted actions:read, 422'd, and fired a false "SECURITY: failed to downscope" error on every approved workflow push before the core fallback recovered. The guard now captures the run's recorded scope before elevating (via the new get_recorded_proxy_permissions) and restores exactly that, falling back to the guaranteed core scope only when the baseline restore fails. Classify installation-token mint failures. get_github_app_installation_token_ with_expiry now treats HTTP 422 (a permission the installation hasn't granted) as the ladder's expected descend signal and keeps it at debug, while a non-422 failure (network/5xx/timeout) is surfaced at WARNING even when errors are otherwise suppressed — so a transient blip no longer silently downscopes a whole run under a debug-only trace. The reduced-scope warning no longer asserts a missing grant as the sole cause. * chore(triage): mark upstream #1701 landed on this branch Ported via PR #181 as Option A (workflows:write kept out of the standing proxy-token scope). Regenerated triage.md from triage.jsonl. * fix: restructure PR creation to POST-first with diagnose-on-failure Move preflight checks from an authoritative gate (before POST) to a diagnostic run after POST failure. This avoids false-positive failures when a just-pushed head branch is momentarily invisible to GitHub ref endpoints, and eliminates 2-3 extra serial API round-trips on the happy path. Also drop unused _PR_CREATED_FALSE indirection and add a docstring to pr_creation_guard acknowledging the fail-open detection design. --------- Co-authored-by: amoussa1229 <166072409+amoussa1229@users.noreply.github.com> Co-authored-by: Ramon Nogueira <ramon.nogueira@langchain.dev> Co-authored-by: Adam Moussa <adam@seahavenind.com>
76 lines
2.7 KiB
Python
76 lines
2.7 KiB
Python
from __future__ import annotations
|
|
|
|
import json
|
|
from typing import Any
|
|
|
|
from langchain_core.messages import ToolMessage
|
|
|
|
from agent.middleware.pr_creation_guard import (
|
|
PullRequestCreationGuardMiddleware,
|
|
is_pr_creation_fallback_command,
|
|
)
|
|
|
|
|
|
class _Request:
|
|
def __init__(self, command: str) -> None:
|
|
self.tool_call = {
|
|
"name": "execute",
|
|
"args": {"command": command},
|
|
"id": "call-1",
|
|
}
|
|
|
|
|
|
async def _handler(_request: Any) -> ToolMessage:
|
|
return ToolMessage(content="allowed", tool_call_id="call-1")
|
|
|
|
|
|
def test_detects_pr_creation_fallback_commands() -> None:
|
|
assert is_pr_creation_fallback_command("GH_TOKEN=dummy gh pr create --draft")
|
|
assert is_pr_creation_fallback_command(
|
|
"gh api repos/langchain-ai/open-swe/pulls -X POST -f title=x"
|
|
)
|
|
assert is_pr_creation_fallback_command(
|
|
"gh api -X POST repos/langchain-ai/open-swe/pulls -f title=x"
|
|
)
|
|
assert is_pr_creation_fallback_command(
|
|
"GH_TOKEN=dummy gh api -X POST repos/langchain-ai/open-swe/pulls -f title=x"
|
|
)
|
|
assert is_pr_creation_fallback_command(
|
|
"curl -X POST https://api.github.com/repos/langchain-ai/open-swe/pulls -d '{}'"
|
|
)
|
|
|
|
|
|
def test_allows_safe_pr_commands() -> None:
|
|
assert not is_pr_creation_fallback_command("GH_TOKEN=dummy gh pr view 1 --json url")
|
|
assert not is_pr_creation_fallback_command("gh pr list --head open-swe/foo")
|
|
assert not is_pr_creation_fallback_command("gh pr edit 1 --add-label ready")
|
|
assert not is_pr_creation_fallback_command("gh pr comment 1 --body done")
|
|
|
|
|
|
async def test_middleware_blocks_execute_pr_creation_fallbacks() -> None:
|
|
for command in (
|
|
"GH_TOKEN=dummy gh pr create --draft",
|
|
"gh api repos/langchain-ai/open-swe/pulls -X POST -f title=x",
|
|
"GH_TOKEN=dummy gh api -X POST repos/langchain-ai/open-swe/pulls -f title=x",
|
|
"curl -X POST https://api.github.com/repos/langchain-ai/open-swe/pulls -d '{}'",
|
|
):
|
|
result = await PullRequestCreationGuardMiddleware().awrap_tool_call(
|
|
_Request(command), _handler
|
|
)
|
|
|
|
assert isinstance(result, ToolMessage)
|
|
assert result.status == "error"
|
|
payload = json.loads(str(result.content))
|
|
assert payload["code"] == "pr_creation_fallback_blocked"
|
|
assert payload["recoverable_by_agent"] is False
|
|
assert "open_pull_request" in payload["error"]
|
|
assert payload["blocked_command"] == command
|
|
|
|
|
|
async def test_middleware_allows_safe_pr_view() -> None:
|
|
result = await PullRequestCreationGuardMiddleware().awrap_tool_call(
|
|
_Request("GH_TOKEN=dummy gh pr view 1 --json url"), _handler
|
|
)
|
|
|
|
assert isinstance(result, ToolMessage)
|
|
assert result.content == "allowed"
|