feat(open-swe): port upstream clean batch (#1788, #1786, #1764, #1782, #1791, #1799) + guard hardening (#226)
Some checks failed
CI / Lint (push) Has been cancelled
CI / Format check (push) Has been cancelled
CI / Typecheck (push) Has been cancelled
CI / Unit tests (push) Has been cancelled
CI / Playwright E2E (push) Has been cancelled
CI / Docker build smoke (push) Has been cancelled
CI / Triage ledger up to date (push) Has been cancelled
CI / ui bun.lock in sync (push) Has been cancelled

* Fix: Fix Insecure Direct Object Reference in slack_start_new_thread.py (#1788)

Co-authored-by: corridor-security[bot] <203152403+corridor-security[bot]@users.noreply.github.com>
(cherry picked from commit 32e81f2979a7baf11fe387df59f7d13a31889c74)

* Fix PR creation guard shell bypasses (#1786)

Co-authored-by: langsmith-fleet[bot] <langsmith-fleet[bot]@users.noreply.github.com>
(cherry picked from commit 75fb8b487852003916c4984504a13ee7226b2ceb)

* fix: add exc_info to swallowed exception in push re-review webhook (#1764)

(cherry picked from commit ab85b372b4f37b7feb849054553daed10852a42c)

* chore: clarify shared response image guidance (#1782)

Co-authored-by: Ramon Nogueira <270434257+ramon-langchain@users.noreply.github.com>
Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
(cherry picked from commit 2e8ff4b72f1148bb36c0c1181063a3abd78b15d0)

* fix: match embedded review description background (#1791)

Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
(cherry picked from commit 4ea2441ada1229bc414b02950d821786be2f7301)

* fix: show current shared thread in sidebar (#1799)

* fix: show current shared thread in sidebar

Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>

* fix: preserve resolved active sidebar threads

Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>

---------

Co-authored-by: Ramon Nogueira <270434257+ramon-langchain@users.noreply.github.com>
Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
Co-authored-by: Johannes du Plessis <51395795+johannes117@users.noreply.github.com>
(cherry picked from commit a77c4e475643b4a55bb2f0c93c0aa2669014fbac)

* chore: switch deferred items to landed in upstream-sync triage documentation and jsonl entries (fork PR #226).

Signed-off-by: Adam Moussa <adam@seahavenind.com>

* harden PR guards + sidebar after security review

- Mirror upstream #1786's nested-shell / executable-normalization hardening
  into the fork-only pr_verdict_guard.py (verdict-gating is a real fork
  control), keeping it in parity with pr_creation_guard.py.
- Close the glued short-flag bypass (bash -c'...') in BOTH guards: a shell's
  -c argument can be concatenated into the same argv token, which the
  space-separated -c detection missed. Diverges pr_creation_guard.py from
  upstream #1786 by design; to be upstreamed.
- Gate the new #1799 sidebar active-thread refresh on ownership so a non-owner
  viewing a shared thread reads last-known state without persisting a metadata
  write (mirrors the is_owner gate on the single-thread read path).
- Fix an F821 in the #1799 cherry-pick (Mapping import / concrete dict type).

Guards remain intentionally fail-open per the honest-agent threat model;
docstrings narrowed to name the residual exotic-shell / stdin-fed vectors.

---------

Signed-off-by: Adam Moussa <adam@seahavenind.com>
Co-authored-by: corridor-security[bot] <203152403+corridor-security[bot]@users.noreply.github.com>
Co-authored-by: John Kennedy <65985482+jkennedyvz@users.noreply.github.com>
Co-authored-by: langsmith-fleet[bot] <langsmith-fleet[bot]@users.noreply.github.com>
Co-authored-by: Suraj Bayas <surajyou24@gmail.com>
Co-authored-by: Ramon Nogueira <ramon.nogueira@langchain.dev>
Co-authored-by: Ramon Nogueira <270434257+ramon-langchain@users.noreply.github.com>
Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
Co-authored-by: Johannes du Plessis <johannes@langchain.dev>
Co-authored-by: Johannes du Plessis <51395795+johannes117@users.noreply.github.com>
This commit is contained in:
Adam Moussa 2026-07-24 18:49:53 -04:00 • committed by GitHub
parent ac773482a2
commit d48cb12e08
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
18 changed files with 535 additions and 34 deletions

View file

@ -1553,6 +1553,7 @@ async def api_list_threads(
async def api_list_threads_sidebar(
active_limit: int = 50,
resolved_limit: int = 20,
active_thread_id: str | None = None,
all: bool = False,
session: dict[str, Any] = _SESSION_DEP,
) -> dict[str, Any]:
@ -1563,6 +1564,7 @@ async def api_list_threads_sidebar(
email=session.get("email"),
active_limit=active_limit,
resolved_limit=resolved_limit,
active_thread_id=active_thread_id,
include_all=all,
)

View file

@ -8,7 +8,7 @@ import binascii
import json
import logging
import os
from collections.abc import AsyncIterator
from collections.abc import AsyncIterator, Mapping
from datetime import UTC, datetime
from typing import Any
@ -722,12 +722,58 @@ async def list_dashboard_threads(
return page["items"]
async def _sidebar_active_thread_summary(
client: Any,
active_thread_id: str | None,
*,
fallback_threads: Mapping[str, dict[str, Any]],
visible_thread_ids: set[str],
login: str,
email: str | None,
include_all: bool,
) -> tuple[dict[str, Any], bool] | None:
if not active_thread_id or active_thread_id in visible_thread_ids:
return None
thread = fallback_threads.get(active_thread_id)
if thread is None:
try:
fetched = await client.threads.get(active_thread_id)
except Exception: # noqa: BLE001
logger.debug(
"Could not fetch active sidebar thread %s", active_thread_id, exc_info=True
)
return None
if not isinstance(fetched, Mapping):
return None
thread = fetched
metadata = _thread_metadata(thread)
if not include_all:
try:
_assert_thread_readable(metadata)
except HTTPException:
return None
# Refreshing persists latest-run metadata back to the thread; only do that
# when the caller owns it, mirroring the is_owner gate on the single-thread
# read path. A non-owner viewing a shared thread reads its last-known state
# without mutating it.
owns_thread = _user_owns_thread(metadata, login, email)
summary = await _summarize_thread(
client,
thread,
owner_login=None if include_all else login,
owner_email=None if include_all else email,
refresh_active_run=owns_thread,
)
return summary, _is_thread_resolved(metadata)
async def list_dashboard_threads_sidebar(
login: str,
*,
email: str | None = None,
active_limit: int = 50,
resolved_limit: int = 20,
active_thread_id: str | None = None,
include_all: bool = False,
) -> dict[str, Any]:
client = langgraph_client()
@ -775,7 +821,8 @@ async def list_dashboard_threads_sidebar(
resolved_candidates = sorted(resolved_threads.values(), key=_thread_updated_ms, reverse=True)
active_window = active_candidates[:safe_active_limit]
resolved_window = resolved_candidates[:safe_resolved_limit]
active_items, resolved_items = await asyncio.gather(
active_ids = {thread_id for thread in active_window if (thread_id := _thread_id(thread))}
active_items, resolved_items, active_thread = await asyncio.gather(
_summarize_threads(
client,
active_window,
@ -788,17 +835,52 @@ async def list_dashboard_threads_sidebar(
owner_login=None if include_all else login,
owner_email=None if include_all else email,
),
_sidebar_active_thread_summary(
client,
active_thread_id,
fallback_threads={**active, **resolved_threads},
visible_thread_ids=active_ids,
login=login,
email=email,
include_all=include_all,
),
)
active_has_more = len(active_candidates) > safe_active_limit
resolved_has_more = len(resolved_candidates) > safe_resolved_limit
if active_thread:
active_thread_summary, is_resolved_active_thread = active_thread
if is_resolved_active_thread:
resolved_items = [
active_thread_summary,
*[item for item in resolved_items if item["id"] != active_thread_summary["id"]],
]
active_items = [
item for item in active_items if item["id"] != active_thread_summary["id"]
]
if len(resolved_items) > safe_resolved_limit:
resolved_items = resolved_items[:safe_resolved_limit]
resolved_has_more = True
else:
active_items = [
active_thread_summary,
*[item for item in active_items if item["id"] != active_thread_summary["id"]],
]
resolved_items = [
item for item in resolved_items if item["id"] != active_thread_summary["id"]
]
if len(active_items) > safe_active_limit:
active_items = active_items[:safe_active_limit]
active_has_more = True
return {
"active": {
"items": active_items,
"limit": safe_active_limit,
"hasMore": len(active_candidates) > safe_active_limit,
"hasMore": active_has_more,
},
"resolved": {
"items": resolved_items,
"limit": safe_resolved_limit,
"hasMore": len(resolved_candidates) > safe_resolved_limit,
"hasMore": resolved_has_more,
},
}

View file

@ -3,6 +3,7 @@
from __future__ import annotations
import json
import os
import re
import shlex
from collections.abc import Awaitable, Callable, Mapping
@ -14,6 +15,9 @@ from langgraph.prebuilt.tool_node import ToolCallRequest
from langgraph.types import Command
_SHELL_SEPARATORS = {";", "&&", "||", "|", "&"}
_SHELL_EXECUTABLES = {"bash", "dash", "sh", "zsh"}
_MAX_SHELL_EXPANSION_DEPTH = 3
_SHELL_EXPANSION_DEPTH_LIMIT_TOKEN = "__pr_creation_guard_shell_expansion_depth_limit__"
_GITHUB_PULLS_ENDPOINT = re.compile(r"(?:^|/)repos/[^/\s]+/[^/\s]+/pulls/?$")
_GITHUB_PULLS_URL = re.compile(r"https://api\.github\.com/repos/[^/\s]+/[^/\s]+/pulls/?")
_BLOCK_ERROR = (
@ -46,13 +50,62 @@ def _tool_call_id(request: ToolCallRequest) -> str | None:
return None
def _shell_tokens(command: str) -> list[str]:
def _split_shell_tokens(command: str) -> list[str]:
try:
return shlex.split(command, posix=True)
except ValueError:
return command.split()
def _executable_name(token: str) -> str:
return os.path.basename(token.strip("'\""))
def _shell_command_argument(tokens: list[str], shell_index: int) -> str | None:
for index, token in enumerate(tokens[shell_index + 1 :], start=shell_index + 1):
if token in _SHELL_SEPARATORS:
return None
if token == "-c" or (
token.startswith("-") and not token.startswith("--") and "c" in token[1:]
):
_, _, glued = token[1:].partition("c")
if glued:
return glued
if index + 1 < len(tokens) and tokens[index + 1] not in _SHELL_SEPARATORS:
return tokens[index + 1]
return None
return None
def _has_nested_shell_command(tokens: list[str]) -> bool:
return any(
_executable_name(token) in _SHELL_EXECUTABLES
and _shell_command_argument(tokens, index) is not None
for index, token in enumerate(tokens)
)
def _expand_nested_shell_tokens(tokens: list[str], depth: int = 0) -> list[str]:
if depth >= _MAX_SHELL_EXPANSION_DEPTH:
if _has_nested_shell_command(tokens):
return [*tokens, _SHELL_EXPANSION_DEPTH_LIMIT_TOKEN]
return tokens
expanded = list(tokens)
for index, token in enumerate(tokens):
if _executable_name(token) not in _SHELL_EXECUTABLES:
continue
inner_command = _shell_command_argument(tokens, index)
if inner_command is None:
continue
expanded.extend(_expand_nested_shell_tokens(_split_shell_tokens(inner_command), depth + 1))
return expanded
def _shell_tokens(command: str) -> list[str]:
return _expand_nested_shell_tokens(_split_shell_tokens(command))
def _is_assignment(token: str) -> bool:
name, sep, _value = token.partition("=")
return bool(sep and name and re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", name))
@ -69,7 +122,7 @@ def _gh_subtokens(tokens: list[str], index: int) -> list[str]:
def _contains_gh_pr_create(tokens: list[str]) -> bool:
for index, token in enumerate(tokens):
if token != "gh":
if _executable_name(token) != "gh":
continue
subtokens = _gh_subtokens(tokens, index)
for offset, subtoken in enumerate(subtokens[:-1]):
@ -136,7 +189,7 @@ def _gh_api_uses_post_or_body(subtokens: list[str]) -> bool:
def _contains_gh_api_pull_create(tokens: list[str]) -> bool:
for index, token in enumerate(tokens):
if token != "gh":
if _executable_name(token) != "gh":
continue
subtokens = _gh_subtokens(tokens, index)
if "api" not in subtokens:
@ -153,7 +206,7 @@ def _contains_gh_api_pull_create(tokens: list[str]) -> bool:
def _contains_direct_pull_create(tokens: list[str]) -> bool:
for index, token in enumerate(tokens):
if token != "curl":
if _executable_name(token) != "curl":
continue
subtokens: list[str] = []
for candidate in tokens[index + 1 :]:
@ -193,7 +246,8 @@ def is_pr_creation_fallback_command(command: str) -> bool:
"""
tokens = _shell_tokens(command)
return (
_contains_gh_pr_create(tokens)
_SHELL_EXPANSION_DEPTH_LIMIT_TOKEN in tokens
or _contains_gh_pr_create(tokens)
or _contains_gh_api_pull_create(tokens)
or _contains_direct_pull_create(tokens)
)

View file

@ -3,6 +3,7 @@
from __future__ import annotations
import json
import os
import re
import shlex
from collections.abc import Awaitable, Callable, Mapping
@ -14,6 +15,9 @@ from langgraph.prebuilt.tool_node import ToolCallRequest
from langgraph.types import Command
_SHELL_SEPARATORS = {";", "&&", "||", "|", "&"}
_SHELL_EXECUTABLES = {"bash", "dash", "sh", "zsh"}
_MAX_SHELL_EXPANSION_DEPTH = 3
_SHELL_EXPANSION_DEPTH_LIMIT_TOKEN = "__pr_verdict_guard_shell_expansion_depth_limit__"
_GITHUB_REVIEWS_ENDPOINT = re.compile(
r"(?:^|/)repos/[^/\s]+/[^/\s]+/pulls/\d+/reviews(?:/\d+/events)?/?$"
)
@ -55,13 +59,62 @@ def _tool_call_id(request: ToolCallRequest) -> str | None:
return None
def _shell_tokens(command: str) -> list[str]:
def _split_shell_tokens(command: str) -> list[str]:
try:
return shlex.split(command, posix=True)
except ValueError:
return command.split()
def _executable_name(token: str) -> str:
return os.path.basename(token.strip("'\""))
def _shell_command_argument(tokens: list[str], shell_index: int) -> str | None:
for index, token in enumerate(tokens[shell_index + 1 :], start=shell_index + 1):
if token in _SHELL_SEPARATORS:
return None
if token == "-c" or (
token.startswith("-") and not token.startswith("--") and "c" in token[1:]
):
_, _, glued = token[1:].partition("c")
if glued:
return glued
if index + 1 < len(tokens) and tokens[index + 1] not in _SHELL_SEPARATORS:
return tokens[index + 1]
return None
return None
def _has_nested_shell_command(tokens: list[str]) -> bool:
return any(
_executable_name(token) in _SHELL_EXECUTABLES
and _shell_command_argument(tokens, index) is not None
for index, token in enumerate(tokens)
)
def _expand_nested_shell_tokens(tokens: list[str], depth: int = 0) -> list[str]:
if depth >= _MAX_SHELL_EXPANSION_DEPTH:
if _has_nested_shell_command(tokens):
return [*tokens, _SHELL_EXPANSION_DEPTH_LIMIT_TOKEN]
return tokens
expanded = list(tokens)
for index, token in enumerate(tokens):
if _executable_name(token) not in _SHELL_EXECUTABLES:
continue
inner_command = _shell_command_argument(tokens, index)
if inner_command is None:
continue
expanded.extend(_expand_nested_shell_tokens(_split_shell_tokens(inner_command), depth + 1))
return expanded
def _shell_tokens(command: str) -> list[str]:
return _expand_nested_shell_tokens(_split_shell_tokens(command))
def _subtokens_after(tokens: list[str], index: int) -> list[str]:
subtokens: list[str] = []
for token in tokens[index + 1 :]:
@ -73,7 +126,7 @@ def _subtokens_after(tokens: list[str], index: int) -> list[str]:
def _contains_gh_pr_review_verdict(tokens: list[str]) -> bool:
for index, token in enumerate(tokens):
if token != "gh":
if _executable_name(token) != "gh":
continue
subtokens = _subtokens_after(tokens, index)
is_pr_review = any(
@ -92,7 +145,7 @@ def _contains_gh_pr_review_verdict(tokens: list[str]) -> bool:
def _contains_gh_api_review_verdict(tokens: list[str]) -> bool:
for index, token in enumerate(tokens):
if token != "gh":
if _executable_name(token) != "gh":
continue
subtokens = _subtokens_after(tokens, index)
if "api" not in subtokens:
@ -111,7 +164,7 @@ def _contains_gh_api_review_verdict(tokens: list[str]) -> bool:
def _contains_direct_review_verdict(tokens: list[str]) -> bool:
for index, token in enumerate(tokens):
if token != "curl":
if _executable_name(token) != "curl":
continue
subtokens = _subtokens_after(tokens, index)
if not any(_GITHUB_REVIEWS_URL.search(subtoken) for subtoken in subtokens):
@ -129,7 +182,13 @@ def is_pr_verdict_fallback_command(command: str) -> bool:
``event=APPROVE|REQUEST_CHANGES`` to a ``/pulls/N/reviews`` endpoint, and
``curl`` to the reviews URL with a verdict event in the body) but is
intentionally fail-open: shell aliases, ``gh`` aliases, ``--input``
JSON-file bodies, and non-curl HTTP clients will not be blocked. This is
JSON-file bodies, and non-curl HTTP clients will not be blocked. The
``-c`` form of a known shell (``bash``/``dash``/``sh``/``zsh``, whether
space-separated or glued as ``-c'...'``) is expanded to a fixed depth and
quoted/path-prefixed executables are normalized, so those do not slip
through; other shells (``ash``/``busybox``/``fish``) and stdin-fed
programs (``... | bash``, here-strings, ``bash -s``) remain fail-open.
This is
acceptable because the threat model is an honest agent papering over a
``publish_review`` limitation or refusal, not an adversary trying to
bypass the guardrail. Plain ``gh pr review`` (no verdict flag) and
@ -138,7 +197,8 @@ def is_pr_verdict_fallback_command(command: str) -> bool:
"""
tokens = _shell_tokens(command)
return (
_contains_gh_pr_review_verdict(tokens)
_SHELL_EXPANSION_DEPTH_LIMIT_TOKEN in tokens
or _contains_gh_pr_review_verdict(tokens)
or _contains_gh_api_review_verdict(tokens)
or _contains_direct_review_verdict(tokens)
)

View file

@ -43,7 +43,10 @@ async def save_plan(
shared content.
Write the content in standard Markdown — headings, bullet/numbered lists, and
fenced code blocks all render.
fenced code blocks all render. Shared responses persist Markdown text only;
they do not upload or serve sandbox-local or relative image files. If images
or screenshots are needed for a Slack response, post them directly in Slack
instead of relying on the shared response page.
Args:
plan_file_path: Path to the Markdown plan file in the sandbox.

View file

@ -2,9 +2,11 @@ import os
import re
from typing import Any
from fastapi import HTTPException
from langgraph.config import get_config
from langgraph_sdk import get_client
from ..dashboard.repo_access import require_repo_access_for_user
from ..dispatch import dispatch_agent_run
from ..utils.dashboard_links import dashboard_thread_url
from ..utils.slack import (
@ -13,6 +15,7 @@ from ..utils.slack import (
store_slack_run_mapping,
)
from ..utils.thread_ids import generate_thread_id_from_slack_thread
from ..webhooks.common import _is_repo_allowed
LANGGRAPH_URL = os.environ.get("LANGGRAPH_URL") or os.environ.get(
"LANGGRAPH_URL_PROD", "http://localhost:2024"
@ -160,6 +163,42 @@ async def slack_start_new_thread(
"error": "default_repo must be a simple owner/name repository string",
}
if default_repo and default_repo.strip() and repo is not None:
if not _is_repo_allowed(repo):
return {
"success": False,
"error": (
f"Repository {repo['owner']}/{repo['name']} is not on the deployment allowlist"
),
}
github_login = configurable.get("github_login")
if not isinstance(github_login, str) or not github_login.strip():
return {
"success": False,
"error": (
"Cannot verify access to the requested repository: no github_login on the "
"parent thread"
),
}
try:
await require_repo_access_for_user(
github_login.strip(), f"{repo['owner']}/{repo['name']}"
)
except HTTPException as exc:
return {
"success": False,
"error": (
f"Access to repository {repo['owner']}/{repo['name']} denied: {exc.detail}"
),
}
except Exception as exc: # noqa: BLE001
return {
"success": False,
"error": (
f"Failed to verify access to repository {repo['owner']}/{repo['name']}: {exc}"
),
}
message_ts, slack_error = await post_slack_top_level_message_with_ts(
channel_id.strip(),
_visible_message(clean_title, clean_instructions, repo),

View file

@ -589,7 +589,9 @@ async def process_github_push_event(payload: dict[str, Any]) -> None:
await common.reconcile_findings_with_review_threads(thread_id, threads)
except Exception:
common.logger.warning(
"Could not sync review threads before push re-review for %s", thread_id
"Could not sync review threads before push re-review for %s",
thread_id,
exc_info=True,
)
pr_meta: ReviewerPRMeta = {

View file

@ -130,14 +130,14 @@
{"sha": "f0897479", "pr": 1778, "subject": "feat: surface context window usage in agents UI (#1778)", "disposition": "deferred", "reason": "Near-clean: context-window indicator UI is all new files (zero drift); options.py hunk must be re-keyed to the fork's Bedrock/Fireworks model map (fork model IDs differ from upstream's) — add context_window per fork entry rather than taking upstream values.", "branch": "context-usage-ui", "local_sha": null, "updated": "2026-07-17T22:33:46Z"}
{"sha": "31263f83", "pr": 1785, "subject": "Fix: Fix Stored XSS in ReplyCard.tsx (#1785)", "disposition": "landed", "reason": "SECURITY priority, near-clean: diff is urlencode() hardening of the GitHub OAuth authorize URL in dashboard routes.py auth_login (upstream bot title says ReplyCard.tsx — mismatch, trust the diff). Fork auth_login has the identical f-string URL; hunk applies clean. Port promptly.", "branch": "feature/upstream-clean-batch-security-linear", "local_sha": null, "updated": "2026-07-20T19:46:00Z"}
{"sha": "3ea29d3f", "pr": 1789, "subject": "Fix: Fix Improper privilege management in server.py (#1789)", "disposition": "landed", "reason": "SECURITY priority, near-clean: adds empty-signature reject to verify_github_signature (utils/github_comments.py) + verify_linear_signature (webhooks/common.py) (title says server.py — mismatch, trust the diff). Both fork functions match at the hunk sites; fork-only verify_jira_secret already guards empty token. Real value: None signature currently raises TypeError in compare_digest. Port BOTH hunks together.", "branch": "feature/upstream-clean-batch-security-linear", "local_sha": null, "updated": "2026-07-20T19:46:00Z"}
{"sha": "2e8ff4b7", "pr": 1782, "subject": "chore: clarify shared response image guidance (#1782)", "disposition": "deferred", "reason": "Near-clean: docstring-only guidance in save_plan (shared responses persist Markdown only, never sandbox-local images — post screenshots directly to Slack) + 2 assertions in test_plan_review. Merges clean against fork. Batch with the next clean-port round.", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-21T18:00:13Z"}
{"sha": "4ea2441a", "pr": 1791, "subject": "fix: match embedded review description background (#1791)", "disposition": "deferred", "reason": "Near-clean UI-only: ReviewMainBody.tsx background match for the embedded review description; merges clean, zero fork drift. Batch with the next clean-port round.", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-21T18:00:13Z"}
{"sha": "2e8ff4b7", "pr": 1782, "subject": "chore: clarify shared response image guidance (#1782)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"}
{"sha": "4ea2441a", "pr": 1791, "subject": "fix: match embedded review description background (#1791)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"}
{"sha": "589dfd83", "pr": 1787, "subject": "fix: Harden Stagehand browser URL handling (#1787)", "disposition": "wont-merge", "reason": "Follows #1648 (wont-merge): fork does not carry the Stagehand browser subagent — this 'fix' re-adds the entire module (992 insertions; modify/delete vs HEAD) plus the stagehand dep in pyproject/uv.lock. Adopting would resurrect a feature rejected under the curated-tools policy.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:14Z"}
{"sha": "9bbd65d3", "pr": 1781, "subject": "fix: derive model context windows from LangChain profiles (#1781)", "disposition": "deferred", "reason": "Pair with deferred #1778 on context-usage-ui: derives model context windows from LangChain profiles. options.py conflicts — re-key to the fork's Bedrock/Fireworks model map (upstream IDs differ; verify LangChain profiles even resolve for the fork's Bedrock-style IDs, else keep static values). uv.lock bumps langchain to a profiles-capable version; test_model_fallback_resolution also conflicts. Port with/after #1778.", "branch": "context-usage-ui", "local_sha": null, "updated": "2026-07-21T18:00:14Z"}
{"sha": "df743658", "pr": 1774, "subject": "feat: add repository skill support to the coding agent (#1774)", "disposition": "wont-merge", "reason": "Upstream reverted this feature in #1800: its factory-time ensure_sandbox_for_thread call ran outside the serialized run and raced the deterministic sandbox name (prod sandbox-create 409s spiking from the #1774 merge date). Feature withdrawn upstream; the fork's deferred security review is moot. If upstream re-lands repo skills (host-side .agents/skills resolution per #1800), triage that new commit fresh — the prompt-injection / trusted-ref concerns in the old reason still apply.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"}
{"sha": "75fb8b48", "pr": 1786, "subject": "Fix PR creation guard shell bypasses (#1786)", "disposition": "deferred", "reason": "SECURITY priority, clean merge: closes shell-bypass holes in PullRequestCreationGuardMiddleware (nested 'bash -c' expansion to depth 3, quoted-executable normalization) — a guard this fork actively wires. Port promptly; ALSO mirror the nested-shell expansion into the fork-only PullRequestVerdictGuardMiddleware (pr_verdict_guard.py), which shares the naive shlex approach and has the same bypass shape.", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-21T18:00:23Z"}
{"sha": "32e81f29", "pr": 1788, "subject": "Fix: Fix Insecure Direct Object Reference in slack_start_new_thread.py (#1788)", "disposition": "deferred", "reason": "SECURITY priority, clean merge: IDOR fix — slack_start_new_thread now enforces the deployment allowlist (_is_repo_allowed) and per-user repo access (require_repo_access_for_user keyed on the parent thread's github_login) before dispatching a run against a different repo. Both helpers exist in the fork at the same paths; merges clean. Port promptly.", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-21T18:00:23Z"}
{"sha": "ab85b372", "pr": 1764, "subject": "fix: add exc_info to swallowed exception in push re-review webhook (#1764)", "disposition": "deferred", "reason": "Trivial: adds exc_info=True to the swallowed exception log in process_github_push_event (webhooks/github.py). Merges clean. Batch with the next clean-port round.", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-21T18:00:23Z"}
{"sha": "75fb8b48", "pr": 1786, "subject": "Fix PR creation guard shell bypasses (#1786)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"}
{"sha": "32e81f29", "pr": 1788, "subject": "Fix: Fix Insecure Direct Object Reference in slack_start_new_thread.py (#1788)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"}
{"sha": "ab85b372", "pr": 1764, "subject": "fix: add exc_info to swallowed exception in push re-review webhook (#1764)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"}
{"sha": "7312851d", "pr": 1777, "subject": "feat: add explicit plan approval tool (#1777)", "disposition": "deferred", "reason": "Feature: explicit approve_plan tool + plan-mode exit. All deps exist in fork (plan_store PLAN_STATUS_APPROVED/SHARED, thread_api._user_owns_thread). Conflicts: prompt.py (fork prompt constants), server.py (tool/middleware wiring), check_message_queue.py (fork dashboard-handoff + mid-run injection drift), AGENTS.md. Hand re-key those four; keep the whole vertical (tool + plan_mode + thread_api + 4 test files) on one side.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:33Z"}
{"sha": "e51abe14", "pr": 1754, "subject": "fix: skip oversized images before model calls (#1754)", "disposition": "deferred", "reason": "Desirable: 10MB image size cap + 'image omitted' text block before model calls, prevents oversized-image model failures. Conflicts only because the fork hardened fetch_image_block (SSRF redirect guard, host-only logging) — hunks are compatible; re-key the size check around the fork's guarded fetch and port impl + test together.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:33Z"}
{"sha": "0ed560a5", "pr": 1779, "subject": "fix: settle review checks after run failures (#1779)", "disposition": "deferred", "reason": "Desirable: run-failure completion webhook now settles reviewer check runs left open when the graph dies (_settle_failed_reviewer_check; uses settle_review_check_run + review_check_pending_result, both already in the fork's review/publish.py from the #214/#215 ports). Conflict is fork drift in completion.py (failure-reply customizations); re-key the new helper in and port with its 159-line test.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:33Z"}
@ -146,7 +146,7 @@
{"sha": "bedc57b8", "pr": 1796, "subject": "fix: cap execute output by making SandboxBackendProxy a BaseSandbox (#1796)", "disposition": "deferred", "reason": "Real fix (unbounded execute stdout pulled into the worker → OOM risk) but inapplicable at the fork's deepagents==0.6.12: no capture-offload API exists (ExecuteOffloadResult / execute_accepts_timeout absent; FilesystemMiddleware has no _resolve_capture BaseSandbox gate), so the bug it fixes cannot occur yet. Re-triage together with deferred #1745 (deepagents 0.6.12→0.7.x bump) — landing that bump makes this fix required. Fork's SandboxBackendProxy has diverged (sync-era, bound_repo repo-binding, no reconnect machinery): hand-apply the execute_with_offload/aexecute_with_offload delegation onto the fork proxy rather than cherry-pick.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"}
{"sha": "e1138cf5", "pr": 1797, "subject": "fix: merge concurrent trusted skill refs (#1797)", "disposition": "wont-merge", "reason": "Follow-up fix to #1774's TrustedSkillsMiddleware, which the fork never adopted (row df743658) and upstream itself reverted in #1800 — nothing to apply.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"}
{"sha": "60e7307c", "pr": 1800, "subject": "revert: repository skill support for the coding agent (#1774, #1797) (#1800)", "disposition": "wont-merge", "reason": "Revert of #1774 + #1797; the fork adopted neither, so there is nothing to revert. Upstream's rationale: #1774's factory-time ensure_sandbox_for_thread ran outside the serialized run and raced the thread-deterministic sandbox name (prod create-409s). Fork invariant holds: its single provisioning call sits inside the __creating__-sentinel lifecycle within the interrupt-serialized run.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"}
{"sha": "a77c4e47", "pr": 1799, "subject": "fix: show current shared thread in sidebar (#1799)", "disposition": "deferred", "reason": "Desirable dashboard fix: sidebar now surfaces the currently-open shared thread (org-member 'Open in Web' links) when it falls outside the caller's own list — _sidebar_active_thread_summary + active_thread_id param; fork has the same include_all/org-member read path in thread_api.py. Cherry-picks CLEAN onto dev (merge-tree probe: zero conflicts); port with its thread-api tests + sandbox_id e2e spec + AgentsSidebar/queries UI as one vertical.", "branch": "", "local_sha": null, "updated": "2026-07-24T19:25:32Z"}
{"sha": "a77c4e47", "pr": 1799, "subject": "fix: show current shared thread in sidebar (#1799)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"}
{"sha": "b67215ed", "pr": 1780, "subject": "feat: link originating Slack threads (#1780)", "disposition": "deferred", "reason": "Desirable for the Slack-heavy fork: dispatch records the originating Slack permalink in thread metadata (webhooks/common.py), open_pull_request embeds an 'Originating Slack thread' link in the PR body (gated to private repos), and the sidebar menu links it. Impl files auto-merge (thread_api / open_pull_request / webhooks/common); conflicts are fork drift in prompt.py (customized prompt constants), tests/e2e/harness.py, tests/slack/test_slack_context.py — hand re-key those; keep backend + UI (AgentsSidebar, types) + e2e dashboard spec as one vertical.", "branch": "", "local_sha": null, "updated": "2026-07-24T19:25:37Z"}
{"sha": "7d79d300", "pr": 1804, "subject": "chore(deps): bump deepagents to 0.7.0b1 pin (#1804)", "disposition": "deferred", "reason": "Rides the deferred #1745 deepagents bump: 0.7.0a7→0.7.0b1 pin + wcmatch>=11.0 uv override (langchain-e2b's e2b dep still caps wcmatch<11.0). Fork is pinned deepagents==0.6.12 — and langsmith==0.10.10 already satisfies the new >=0.10.9 floor — and never took the a7-era HARNESS_EXCLUDED_MIDDLEWARE code its prompt.py hunk deletes, so nothing applies standalone. Re-triage as one cluster: #1745 (bump, now targeting 0.7.0b1) + #1796 (offload delegation) + this pin/override.", "branch": "", "local_sha": null, "updated": "2026-07-24T19:25:44Z"}
{"sha": "0e2eefbc", "pr": 1801, "subject": "chore: bump default sandbox resources and delete-after-stop (#1801)", "disposition": "deferred", "reason": "Capacity/cost policy, not a correctness fix: defaults 2→4 vCPU, 7936MiB→16GiB mem, 32→128GiB FS, stopped-sandbox delete 24h→14d. Would materially raise LangSmith sandbox spend on sh-openswe unless prod env already overrides; all five values are env-overridable today. Decide prod sizing first (ops decision), then adopt or reject. Textual conflict is only fork drift in integrations/langsmith.py (sync-era SandboxClient, reconnect constants absent) — the constants re-key trivially; docs hunks auto-merge.", "branch": "", "local_sha": null, "updated": "2026-07-24T19:25:49Z"}

View file

@ -88,6 +88,12 @@ Rows key on the **upstream SHA** (stable across local cherry-picks). Deferred ro
| `b5e52925` | #1775 | feat: add structured Linear issue filters (#1775) | Landed | Clean pick: structured filters for linear_search_issues — stacks directly on #1748 (landed PR #206); zero fork drift on all three files. Ready whenever. | feature/upstream-clean-batch-security-linear |
| `31263f83` | #1785 | Fix: Fix Stored XSS in ReplyCard.tsx (#1785) | Landed | SECURITY priority, near-clean: diff is urlencode() hardening of the GitHub OAuth authorize URL in dashboard routes.py auth_login (upstream bot title says ReplyCard.tsx — mismatch, trust the diff). Fork auth_login has the identical f-string URL; hunk applies clean. Port promptly. | feature/upstream-clean-batch-security-linear |
| `3ea29d3f` | #1789 | Fix: Fix Improper privilege management in server.py (#1789) | Landed | SECURITY priority, near-clean: adds empty-signature reject to verify_github_signature (utils/github_comments.py) + verify_linear_signature (webhooks/common.py) (title says server.py — mismatch, trust the diff). Both fork functions match at the hunk sites; fork-only verify_jira_secret already guards empty token. Real value: None signature currently raises TypeError in compare_digest. Port BOTH hunks together. | feature/upstream-clean-batch-security-linear |
| `2e8ff4b7` | #1782 | chore: clarify shared response image guidance (#1782) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 |
| `4ea2441a` | #1791 | fix: match embedded review description background (#1791) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 |
| `75fb8b48` | #1786 | Fix PR creation guard shell bypasses (#1786) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 |
| `32e81f29` | #1788 | Fix: Fix Insecure Direct Object Reference in slack_start_new_thread.py (#1788) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 |
| `ab85b372` | #1764 | fix: add exc_info to swallowed exception in push re-review webhook (#1764) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 |
| `a77c4e47` | #1799 | fix: show current shared thread in sidebar (#1799) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 |
| `c3292d82` | #1611 | bake sfw binary into sandbox image | Won't merge | already in dev | |
| `48bf712b` | #1609 | show message timestamps | Won't merge | already in dev | |
| `85c0f63e` | #1620 | clickable shared PR header | Won't merge | already in dev | |
@ -147,17 +153,11 @@ Rows key on the **upstream SHA** (stable across local cherry-picks). Deferred ro
| `81d544bc` | #1765 | feat: Expose more specific AGENTS.md context to Open SWE reviewer (#1765) | Deferred | Near-clean: agents_md.py helper + tests are zero-drift; reviewer.py hunk is small (~39 lines) but lands in the fork's heavily-diverged reviewer — hand-apply the scoped_agents_md wiring onto the fork's fetch_agents_md call sites (reviewer.py ~L404/445/1031). | reviewer-context |
| `8c8e58bc` | #1776 | feat: connect automations to Slack channels (#1776) | Deferred | Moderate reconcile: Slack-channel wiring for automations. Fork helpers exist (post_slack_top_level_message_with_ts, generate_thread_id_from_slack_thread, slack_id_for_login); completion.py (+233 drift) and schedules.py (+167) need hand-merge; UI automations feature present. Keep backend+UI+tests as one vertical. | automations-slack |
| `f0897479` | #1778 | feat: surface context window usage in agents UI (#1778) | Deferred | Near-clean: context-window indicator UI is all new files (zero drift); options.py hunk must be re-keyed to the fork's Bedrock/Fireworks model map (fork model IDs differ from upstream's) — add context_window per fork entry rather than taking upstream values. | context-usage-ui |
| `2e8ff4b7` | #1782 | chore: clarify shared response image guidance (#1782) | Deferred | Near-clean: docstring-only guidance in save_plan (shared responses persist Markdown only, never sandbox-local images — post screenshots directly to Slack) + 2 assertions in test_plan_review. Merges clean against fork. Batch with the next clean-port round. | feature/upstream-clean-batch-jul21 |
| `4ea2441a` | #1791 | fix: match embedded review description background (#1791) | Deferred | Near-clean UI-only: ReviewMainBody.tsx background match for the embedded review description; merges clean, zero fork drift. Batch with the next clean-port round. | feature/upstream-clean-batch-jul21 |
| `9bbd65d3` | #1781 | fix: derive model context windows from LangChain profiles (#1781) | Deferred | Pair with deferred #1778 on context-usage-ui: derives model context windows from LangChain profiles. options.py conflicts — re-key to the fork's Bedrock/Fireworks model map (upstream IDs differ; verify LangChain profiles even resolve for the fork's Bedrock-style IDs, else keep static values). uv.lock bumps langchain to a profiles-capable version; test_model_fallback_resolution also conflicts. Port with/after #1778. | context-usage-ui |
| `75fb8b48` | #1786 | Fix PR creation guard shell bypasses (#1786) | Deferred | SECURITY priority, clean merge: closes shell-bypass holes in PullRequestCreationGuardMiddleware (nested 'bash -c' expansion to depth 3, quoted-executable normalization) — a guard this fork actively wires. Port promptly; ALSO mirror the nested-shell expansion into the fork-only PullRequestVerdictGuardMiddleware (pr_verdict_guard.py), which shares the naive shlex approach and has the same bypass shape. | feature/upstream-clean-batch-jul21 |
| `32e81f29` | #1788 | Fix: Fix Insecure Direct Object Reference in slack_start_new_thread.py (#1788) | Deferred | SECURITY priority, clean merge: IDOR fix — slack_start_new_thread now enforces the deployment allowlist (_is_repo_allowed) and per-user repo access (require_repo_access_for_user keyed on the parent thread's github_login) before dispatching a run against a different repo. Both helpers exist in the fork at the same paths; merges clean. Port promptly. | feature/upstream-clean-batch-jul21 |
| `ab85b372` | #1764 | fix: add exc_info to swallowed exception in push re-review webhook (#1764) | Deferred | Trivial: adds exc_info=True to the swallowed exception log in process_github_push_event (webhooks/github.py). Merges clean. Batch with the next clean-port round. | feature/upstream-clean-batch-jul21 |
| `7312851d` | #1777 | feat: add explicit plan approval tool (#1777) | Deferred | Feature: explicit approve_plan tool + plan-mode exit. All deps exist in fork (plan_store PLAN_STATUS_APPROVED/SHARED, thread_api._user_owns_thread). Conflicts: prompt.py (fork prompt constants), server.py (tool/middleware wiring), check_message_queue.py (fork dashboard-handoff + mid-run injection drift), AGENTS.md. Hand re-key those four; keep the whole vertical (tool + plan_mode + thread_api + 4 test files) on one side. | |
| `e51abe14` | #1754 | fix: skip oversized images before model calls (#1754) | Deferred | Desirable: 10MB image size cap + 'image omitted' text block before model calls, prevents oversized-image model failures. Conflicts only because the fork hardened fetch_image_block (SSRF redirect guard, host-only logging) — hunks are compatible; re-key the size check around the fork's guarded fetch and port impl + test together. | |
| `0ed560a5` | #1779 | fix: settle review checks after run failures (#1779) | Deferred | Desirable: run-failure completion webhook now settles reviewer check runs left open when the graph dies (_settle_failed_reviewer_check; uses settle_review_check_run + review_check_pending_result, both already in the fork's review/publish.py from the #214/#215 ports). Conflict is fork drift in completion.py (failure-reply customizations); re-key the new helper in and port with its 159-line test. | |
| `bedc57b8` | #1796 | fix: cap execute output by making SandboxBackendProxy a BaseSandbox (#1796) | Deferred | Real fix (unbounded execute stdout pulled into the worker → OOM risk) but inapplicable at the fork's deepagents==0.6.12: no capture-offload API exists (ExecuteOffloadResult / execute_accepts_timeout absent; FilesystemMiddleware has no _resolve_capture BaseSandbox gate), so the bug it fixes cannot occur yet. Re-triage together with deferred #1745 (deepagents 0.6.12→0.7.x bump) — landing that bump makes this fix required. Fork's SandboxBackendProxy has diverged (sync-era, bound_repo repo-binding, no reconnect machinery): hand-apply the execute_with_offload/aexecute_with_offload delegation onto the fork proxy rather than cherry-pick. | |
| `a77c4e47` | #1799 | fix: show current shared thread in sidebar (#1799) | Deferred | Desirable dashboard fix: sidebar now surfaces the currently-open shared thread (org-member 'Open in Web' links) when it falls outside the caller's own list — _sidebar_active_thread_summary + active_thread_id param; fork has the same include_all/org-member read path in thread_api.py. Cherry-picks CLEAN onto dev (merge-tree probe: zero conflicts); port with its thread-api tests + sandbox_id e2e spec + AgentsSidebar/queries UI as one vertical. | |
| `b67215ed` | #1780 | feat: link originating Slack threads (#1780) | Deferred | Desirable for the Slack-heavy fork: dispatch records the originating Slack permalink in thread metadata (webhooks/common.py), open_pull_request embeds an 'Originating Slack thread' link in the PR body (gated to private repos), and the sidebar menu links it. Impl files auto-merge (thread_api / open_pull_request / webhooks/common); conflicts are fork drift in prompt.py (customized prompt constants), tests/e2e/harness.py, tests/slack/test_slack_context.py — hand re-key those; keep backend + UI (AgentsSidebar, types) + e2e dashboard spec as one vertical. | |
| `7d79d300` | #1804 | chore(deps): bump deepagents to 0.7.0b1 pin (#1804) | Deferred | Rides the deferred #1745 deepagents bump: 0.7.0a7→0.7.0b1 pin + wcmatch>=11.0 uv override (langchain-e2b's e2b dep still caps wcmatch<11.0). Fork is pinned deepagents==0.6.12 — and langsmith==0.10.10 already satisfies the new >=0.10.9 floor — and never took the a7-era HARNESS_EXCLUDED_MIDDLEWARE code its prompt.py hunk deletes, so nothing applies standalone. Re-triage as one cluster: #1745 (bump, now targeting 0.7.0b1) + #1796 (offload delegation) + this pin/override. | |
| `0e2eefbc` | #1801 | chore: bump default sandbox resources and delete-after-stop (#1801) | Deferred | Capacity/cost policy, not a correctness fix: defaults 2→4 vCPU, 7936MiB→16GiB mem, 32→128GiB FS, stopped-sandbox delete 24h→14d. Would materially raise LangSmith sandbox spend on sh-openswe unless prod env already overrides; all five values are env-overridable today. Decide prod sizing first (ops decision), then adopt or reject. Textual conflict is only fork drift in integrations/langsmith.py (sync-era SandboxClient, reconnect constants absent) — the constants re-key trivially; docs hunks auto-merge. | |

View file

@ -214,6 +214,14 @@ def test_save_plan_exported_and_wired() -> None:
assert callable(save_plan)
def test_save_plan_description_warns_about_slack_images() -> None:
from agent.tools import save_plan
description = save_plan.__doc__ or ""
assert "persist Markdown text only" in description
assert "post them directly in Slack" in description
def test_plan_status_constants() -> None:
from agent.dashboard import plan_store

View file

@ -1360,6 +1360,161 @@ async def test_list_dashboard_threads_sidebar_fills_buckets_with_one_endpoint(mo
assert {call["offset"] for call in searches} == {0, page_size}
async def test_list_dashboard_threads_sidebar_includes_readable_active_thread(
monkeypatch,
) -> None:
threads = _make_threads(1, resolved_before=0)
shared_thread = {
"thread_id": "shared-thread",
"metadata": {
"source": "slack",
"github_login": "teammate",
"title": "Teammate thread",
"updated_at_ms": 100,
"latest_run_status": "success",
"sandbox_id": "sandbox-123",
},
}
class FakeThreads:
async def search(self, *, metadata, limit, offset, sort_by, sort_order, select):
assert select == thread_api._THREAD_LIST_SELECT
return threads[offset : offset + limit]
async def get(self, thread_id):
assert thread_id == "shared-thread"
return shared_thread
async def update(self, *, thread_id, metadata):
return None
class FakeRuns:
async def list(self, thread_id, limit=1):
return []
class FakeClient:
threads = FakeThreads()
runs = FakeRuns()
monkeypatch.setattr(thread_api, "langgraph_client", lambda: FakeClient())
result = await thread_api.list_dashboard_threads_sidebar(
"octocat",
email=None,
active_limit=5,
resolved_limit=5,
active_thread_id="shared-thread",
)
assert [item["id"] for item in result["active"]["items"]] == ["shared-thread", "t0"]
shared = result["active"]["items"][0]
assert shared["isOwner"] is False
assert shared["sandboxId"] == "sandbox-123"
async def test_list_dashboard_threads_sidebar_keeps_resolved_active_thread_resolved(
monkeypatch,
) -> None:
threads = _make_threads(1, resolved_before=0)
shared_thread = {
"thread_id": "shared-resolved-thread",
"metadata": {
"source": "slack",
"github_login": "teammate",
"title": "Resolved teammate thread",
"updated_at_ms": 100,
"latest_run_status": "success",
"resolved": True,
"sandbox_id": "sandbox-456",
},
}
class FakeThreads:
async def search(self, *, metadata, limit, offset, sort_by, sort_order, select):
assert select == thread_api._THREAD_LIST_SELECT
return threads[offset : offset + limit]
async def get(self, thread_id):
assert thread_id == "shared-resolved-thread"
return shared_thread
async def update(self, *, thread_id, metadata):
return None
class FakeRuns:
async def list(self, thread_id, limit=1):
return []
class FakeClient:
threads = FakeThreads()
runs = FakeRuns()
monkeypatch.setattr(thread_api, "langgraph_client", lambda: FakeClient())
result = await thread_api.list_dashboard_threads_sidebar(
"octocat",
email=None,
active_limit=5,
resolved_limit=5,
active_thread_id="shared-resolved-thread",
)
assert [item["id"] for item in result["active"]["items"]] == ["t0"]
assert [item["id"] for item in result["resolved"]["items"]] == ["shared-resolved-thread"]
shared = result["resolved"]["items"][0]
assert shared["isOwner"] is False
assert shared["resolved"] is True
assert shared["sandboxId"] == "sandbox-456"
async def test_list_dashboard_threads_sidebar_ignores_unreadable_active_thread(
monkeypatch,
) -> None:
threads = _make_threads(1, resolved_before=0)
private_thread = {
"thread_id": "private-thread",
"metadata": {
"source": "internal",
"github_login": "teammate",
"title": "Private thread",
"updated_at_ms": 100,
"latest_run_status": "success",
},
}
class FakeThreads:
async def search(self, *, metadata, limit, offset, sort_by, sort_order, select):
assert select == thread_api._THREAD_LIST_SELECT
return threads[offset : offset + limit]
async def get(self, thread_id):
assert thread_id == "private-thread"
return private_thread
async def update(self, *, thread_id, metadata):
return None
class FakeRuns:
async def list(self, thread_id, limit=1):
return []
class FakeClient:
threads = FakeThreads()
runs = FakeRuns()
monkeypatch.setattr(thread_api, "langgraph_client", lambda: FakeClient())
result = await thread_api.list_dashboard_threads_sidebar(
"octocat",
email=None,
active_limit=5,
resolved_limit=5,
active_thread_id="private-thread",
)
assert [item["id"] for item in result["active"]["items"]] == ["t0"]
async def test_list_dashboard_threads_page_refreshes_only_unsettled_threads(monkeypatch) -> None:
threads = _make_threads(3, resolved_before=0)
threads[0]["metadata"]["latest_run_status"] = "success"

View file

@ -5,6 +5,7 @@ import { test, expect, type Locator, type Page } from "@playwright/test";
// creates a sandbox and stamps its id into the thread metadata, and the UI
// copies that same id. Only the LLM/GitHub/Slack boundaries are faked.
const SAME_USER = { login: "alice", email: "alice@example.com" };
const OTHER_USER = { login: "bob", email: "bob@example.com" };
async function loginAs(page: Page, user: { login: string; email: string }) {
const res = await page.request.post("/control/login", { data: user });
@ -112,4 +113,39 @@ test.describe("thread sandbox id (real dashboard UI)", () => {
await context.close();
});
test("shared active thread appears in the sidebar with sandbox action", async ({
page,
browser,
baseURL,
}, testInfo) => {
await loginAs(page, SAME_USER);
const threadId = await createThreadWithSandbox(page);
const bobContext = await browser.newContext({ baseURL });
await bobContext.grantPermissions(["clipboard-read", "clipboard-write"], {
origin: baseURL,
});
const bobPage = await bobContext.newPage();
await loginAs(bobPage, OTHER_USER);
await bobPage.goto(`/agents/${threadId}`);
await expect(bobPage).toHaveURL(new RegExp(`/agents/${threadId}$`));
const row = bobPage.locator(`a[href$="/agents/${threadId}"]`).first();
await expect(row).toBeVisible();
await row.hover();
await kebabFor(row).click();
await expect(copyItem(bobPage)).toBeEnabled();
const screenshotPath = testInfo.outputPath(
"shared-thread-sidebar-sandbox-menu.png",
);
await bobPage.screenshot({ path: screenshotPath, fullPage: true });
await testInfo.attach("shared-thread-sidebar-sandbox-menu", {
path: screenshotPath,
contentType: "image/png",
});
await bobContext.close();
});
});

View file

@ -38,6 +38,23 @@ def test_detects_pr_creation_fallback_commands() -> None:
assert is_pr_creation_fallback_command(
"curl -X POST https://api.github.com/repos/langchain-ai/open-swe/pulls -d '{}'"
)
assert is_pr_creation_fallback_command("/usr/bin/gh pr create --draft")
assert is_pr_creation_fallback_command(
"/usr/bin/curl -X POST https://api.github.com/repos/langchain-ai/open-swe/pulls -d '{}'"
)
assert is_pr_creation_fallback_command("bash -c 'gh pr create --draft'")
assert is_pr_creation_fallback_command("bash -c'gh pr create --draft'")
assert is_pr_creation_fallback_command("zsh -lc'gh pr create --draft'")
assert is_pr_creation_fallback_command("GH_TOKEN=dummy sh -c 'gh pr create --draft'")
assert is_pr_creation_fallback_command(
"zsh -lc 'gh api repos/langchain-ai/open-swe/pulls -X POST -f title=x'"
)
assert is_pr_creation_fallback_command(
"bash -c \"curl -X POST https://api.github.com/repos/langchain-ai/open-swe/pulls -d '{}'\""
)
assert is_pr_creation_fallback_command(
'bash -c \'sh -c "dash -c \\"zsh -c \\\\\\"gh pr create --draft\\\\\\"\\""\''
)
def test_allows_safe_pr_commands() -> None:
@ -45,6 +62,8 @@ def test_allows_safe_pr_commands() -> None:
assert not is_pr_creation_fallback_command("gh pr list --head open-swe/foo")
assert not is_pr_creation_fallback_command("gh pr edit 1 --add-label ready")
assert not is_pr_creation_fallback_command("gh pr comment 1 --body done")
assert not is_pr_creation_fallback_command("bash -c 'gh pr view 1 --json url'")
assert not is_pr_creation_fallback_command("/usr/bin/gh pr view 1 --json url")
async def test_middleware_blocks_execute_pr_creation_fallbacks() -> None:

View file

@ -60,6 +60,28 @@ def test_detects_curl_review_verdicts() -> None:
)
def test_detects_nested_and_normalized_verdict_commands() -> None:
assert is_pr_verdict_fallback_command("/usr/bin/gh pr review 12 --approve")
assert is_pr_verdict_fallback_command(
"/usr/bin/curl -X POST https://api.github.com/repos/o/r/pulls/5/reviews "
'-d \'{"event": "APPROVE"}\''
)
assert is_pr_verdict_fallback_command("bash -c 'gh pr review 12 --approve'")
assert is_pr_verdict_fallback_command("bash -c'gh pr review 12 --approve'")
assert is_pr_verdict_fallback_command("zsh -lc'gh pr review 12 --approve'")
assert is_pr_verdict_fallback_command("GH_TOKEN=dummy sh -c 'gh pr review -a'")
assert is_pr_verdict_fallback_command(
"zsh -lc 'gh api repos/o/r/pulls/5/reviews -X POST -f event=REQUEST_CHANGES'"
)
assert is_pr_verdict_fallback_command(
'bash -c "curl -X POST https://api.github.com/repos/o/r/pulls/5/reviews '
'-d \'{\\"event\\": \\"APPROVE\\"}\'"'
)
assert is_pr_verdict_fallback_command(
'bash -c \'sh -c "dash -c \\"zsh -c \\\\\\"gh pr review 12 --approve\\\\\\"\\""\''
)
def test_allows_safe_review_commands() -> None:
assert not is_pr_verdict_fallback_command("gh pr review 12 --comment -b 'looks good'")
assert not is_pr_verdict_fallback_command("gh pr review -c")
@ -69,6 +91,8 @@ def test_allows_safe_review_commands() -> None:
assert not is_pr_verdict_fallback_command("gh api repos/o/r/pulls/5/reviews")
assert not is_pr_verdict_fallback_command("gh pr create --draft")
assert not is_pr_verdict_fallback_command("curl https://api.github.com/repos/o/r/pulls/5")
assert not is_pr_verdict_fallback_command("bash -c 'gh pr review 12 --comment -b ok'")
assert not is_pr_verdict_fallback_command("/usr/bin/gh pr view 12 --json reviews")
async def test_middleware_blocks_execute_verdict_fallbacks() -> None:

View file

@ -111,7 +111,7 @@ export function AgentsSidebar({
}: AgentsSidebarProps) {
const { prefs, setGroup, setCompact, setFilters, resetFilters } =
useSidebarPrefs()
const sidebar = useSidebarThreads(RESOLVED_SIDEBAR_LIMIT)
const sidebar = useSidebarThreads(RESOLVED_SIDEBAR_LIMIT, activeThreadId)
const activeThreads = sidebar.data?.active.items ?? []
const resolvedThreads = sidebar.data?.resolved.items ?? []
const resolvedHasMore = sidebar.data?.resolved.hasMore ?? false

View file

@ -225,6 +225,7 @@ function buildThreadsPageQuery(params: ThreadsPageParams): string {
function buildSidebarThreadsQuery(params: {
activeLimit?: number
resolvedLimit?: number
activeThreadId?: string
}): string {
const search = new URLSearchParams()
if (params.activeLimit != null) {
@ -233,6 +234,9 @@ function buildSidebarThreadsQuery(params: {
if (params.resolvedLimit != null) {
search.set("resolved_limit", String(params.resolvedLimit))
}
if (params.activeThreadId) {
search.set("active_thread_id", params.activeThreadId)
}
const query = search.toString()
return query ? `?${query}` : ""
}
@ -242,6 +246,7 @@ export const agentsApi = {
listSidebarThreads: (params: {
activeLimit?: number
resolvedLimit?: number
activeThreadId?: string
}) =>
agentsRequest<SidebarThreads>(
`/threads/sidebar${buildSidebarThreadsQuery(params)}`

View file

@ -13,8 +13,11 @@ import type { AgentThread, Chunk, ImageChunk, Message } from "./types"
export const agentThreadKeys = {
lists: ["agent-threads", "lists"] as const,
sidebar: (params: { activeLimit: number; resolvedLimit: number }) =>
["agent-threads", "lists", "sidebar", params] as const,
sidebar: (params: {
activeLimit: number
resolvedLimit: number
activeThreadId?: string
}) => ["agent-threads", "lists", "sidebar", params] as const,
detail: (threadId: string) => ["agent-threads", threadId] as const,
prDiff: (threadId: string) => ["agent-threads", threadId, "pr-diff"] as const,
workflowApprovals: (threadId: string) =>
@ -93,10 +96,14 @@ function sidebarRefetchInterval(query: { state: { data?: SidebarThreads } }) {
: false
}
export function useSidebarThreads(resolvedLimit: number) {
export function useSidebarThreads(
resolvedLimit: number,
activeThreadId?: string
) {
const params = {
activeLimit: SIDEBAR_ACTIVE_LIMIT,
resolvedLimit,
activeThreadId,
}
return useQuery({
queryKey: agentThreadKeys.sidebar(params),

View file

@ -1262,7 +1262,12 @@ function ReviewBodyInner({
deletions: detail.pr.deletions,
}}
/>
<div className="mt-4 rounded-lg border border-border bg-card p-4">
<div
className={cn(
"mt-4 rounded-lg border border-border p-4",
embedded ? "bg-[var(--ui-surface)]" : "bg-card"
)}
>
{detail.pr.body ? (
<Markdown
content={detail.pr.body}