feat(open-swe): port upstream clean batch (#1788, #1786, #1764, #1782, #1791, #1799) + guard hardening (#226)
Some checks failed
CI / Lint (push) Has been cancelled
CI / Format check (push) Has been cancelled
CI / Typecheck (push) Has been cancelled
CI / Unit tests (push) Has been cancelled
CI / Playwright E2E (push) Has been cancelled
CI / Docker build smoke (push) Has been cancelled
CI / Triage ledger up to date (push) Has been cancelled
CI / ui bun.lock in sync (push) Has been cancelled

* Fix: Fix Insecure Direct Object Reference in slack_start_new_thread.py (#1788)

Co-authored-by: corridor-security[bot] <203152403+corridor-security[bot]@users.noreply.github.com>
(cherry picked from commit 32e81f2979a7baf11fe387df59f7d13a31889c74)

* Fix PR creation guard shell bypasses (#1786)

Co-authored-by: langsmith-fleet[bot] <langsmith-fleet[bot]@users.noreply.github.com>
(cherry picked from commit 75fb8b487852003916c4984504a13ee7226b2ceb)

* fix: add exc_info to swallowed exception in push re-review webhook (#1764)

(cherry picked from commit ab85b372b4f37b7feb849054553daed10852a42c)

* chore: clarify shared response image guidance (#1782)

Co-authored-by: Ramon Nogueira <270434257+ramon-langchain@users.noreply.github.com>
Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
(cherry picked from commit 2e8ff4b72f1148bb36c0c1181063a3abd78b15d0)

* fix: match embedded review description background (#1791)

Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
(cherry picked from commit 4ea2441ada1229bc414b02950d821786be2f7301)

* fix: show current shared thread in sidebar (#1799)

* fix: show current shared thread in sidebar

Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>

* fix: preserve resolved active sidebar threads

Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>

---------

Co-authored-by: Ramon Nogueira <270434257+ramon-langchain@users.noreply.github.com>
Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
Co-authored-by: Johannes du Plessis <51395795+johannes117@users.noreply.github.com>
(cherry picked from commit a77c4e475643b4a55bb2f0c93c0aa2669014fbac)

* chore: switch deferred items to landed in upstream-sync triage documentation and jsonl entries (fork PR #226).

Signed-off-by: Adam Moussa <adam@seahavenind.com>

* harden PR guards + sidebar after security review

- Mirror upstream #1786's nested-shell / executable-normalization hardening
  into the fork-only pr_verdict_guard.py (verdict-gating is a real fork
  control), keeping it in parity with pr_creation_guard.py.
- Close the glued short-flag bypass (bash -c'...') in BOTH guards: a shell's
  -c argument can be concatenated into the same argv token, which the
  space-separated -c detection missed. Diverges pr_creation_guard.py from
  upstream #1786 by design; to be upstreamed.
- Gate the new #1799 sidebar active-thread refresh on ownership so a non-owner
  viewing a shared thread reads last-known state without persisting a metadata
  write (mirrors the is_owner gate on the single-thread read path).
- Fix an F821 in the #1799 cherry-pick (Mapping import / concrete dict type).

Guards remain intentionally fail-open per the honest-agent threat model;
docstrings narrowed to name the residual exotic-shell / stdin-fed vectors.

---------

Signed-off-by: Adam Moussa <adam@seahavenind.com>
Co-authored-by: corridor-security[bot] <203152403+corridor-security[bot]@users.noreply.github.com>
Co-authored-by: John Kennedy <65985482+jkennedyvz@users.noreply.github.com>
Co-authored-by: langsmith-fleet[bot] <langsmith-fleet[bot]@users.noreply.github.com>
Co-authored-by: Suraj Bayas <surajyou24@gmail.com>
Co-authored-by: Ramon Nogueira <ramon.nogueira@langchain.dev>
Co-authored-by: Ramon Nogueira <270434257+ramon-langchain@users.noreply.github.com>
Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
Co-authored-by: Johannes du Plessis <johannes@langchain.dev>
Co-authored-by: Johannes du Plessis <51395795+johannes117@users.noreply.github.com>
This commit is contained in:
Adam Moussa 2026-07-24 18:49:53 -04:00 • committed by GitHub
parent ac773482a2
commit d48cb12e08
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
18 changed files with 535 additions and 34 deletions

View file

@ -1553,6 +1553,7 @@ async def api_list_threads(
async def api_list_threads_sidebar( async def api_list_threads_sidebar(
active_limit: int = 50, active_limit: int = 50,
resolved_limit: int = 20, resolved_limit: int = 20,
active_thread_id: str | None = None,
all: bool = False, all: bool = False,
session: dict[str, Any] = _SESSION_DEP, session: dict[str, Any] = _SESSION_DEP,
) -> dict[str, Any]: ) -> dict[str, Any]:
@ -1563,6 +1564,7 @@ async def api_list_threads_sidebar(
email=session.get("email"), email=session.get("email"),
active_limit=active_limit, active_limit=active_limit,
resolved_limit=resolved_limit, resolved_limit=resolved_limit,
active_thread_id=active_thread_id,
include_all=all, include_all=all,
) )

View file

@ -8,7 +8,7 @@ import binascii
import json import json
import logging import logging
import os import os
from collections.abc import AsyncIterator from collections.abc import AsyncIterator, Mapping
from datetime import UTC, datetime from datetime import UTC, datetime
from typing import Any from typing import Any
@ -722,12 +722,58 @@ async def list_dashboard_threads(
return page["items"] return page["items"]
async def _sidebar_active_thread_summary(
client: Any,
active_thread_id: str | None,
*,
fallback_threads: Mapping[str, dict[str, Any]],
visible_thread_ids: set[str],
login: str,
email: str | None,
include_all: bool,
) -> tuple[dict[str, Any], bool] | None:
if not active_thread_id or active_thread_id in visible_thread_ids:
return None
thread = fallback_threads.get(active_thread_id)
if thread is None:
try:
fetched = await client.threads.get(active_thread_id)
except Exception: # noqa: BLE001
logger.debug(
"Could not fetch active sidebar thread %s", active_thread_id, exc_info=True
)
return None
if not isinstance(fetched, Mapping):
return None
thread = fetched
metadata = _thread_metadata(thread)
if not include_all:
try:
_assert_thread_readable(metadata)
except HTTPException:
return None
# Refreshing persists latest-run metadata back to the thread; only do that
# when the caller owns it, mirroring the is_owner gate on the single-thread
# read path. A non-owner viewing a shared thread reads its last-known state
# without mutating it.
owns_thread = _user_owns_thread(metadata, login, email)
summary = await _summarize_thread(
client,
thread,
owner_login=None if include_all else login,
owner_email=None if include_all else email,
refresh_active_run=owns_thread,
)
return summary, _is_thread_resolved(metadata)
async def list_dashboard_threads_sidebar( async def list_dashboard_threads_sidebar(
login: str, login: str,
*, *,
email: str | None = None, email: str | None = None,
active_limit: int = 50, active_limit: int = 50,
resolved_limit: int = 20, resolved_limit: int = 20,
active_thread_id: str | None = None,
include_all: bool = False, include_all: bool = False,
) -> dict[str, Any]: ) -> dict[str, Any]:
client = langgraph_client() client = langgraph_client()
@ -775,7 +821,8 @@ async def list_dashboard_threads_sidebar(
resolved_candidates = sorted(resolved_threads.values(), key=_thread_updated_ms, reverse=True) resolved_candidates = sorted(resolved_threads.values(), key=_thread_updated_ms, reverse=True)
active_window = active_candidates[:safe_active_limit] active_window = active_candidates[:safe_active_limit]
resolved_window = resolved_candidates[:safe_resolved_limit] resolved_window = resolved_candidates[:safe_resolved_limit]
active_items, resolved_items = await asyncio.gather( active_ids = {thread_id for thread in active_window if (thread_id := _thread_id(thread))}
active_items, resolved_items, active_thread = await asyncio.gather(
_summarize_threads( _summarize_threads(
client, client,
active_window, active_window,
@ -788,17 +835,52 @@ async def list_dashboard_threads_sidebar(
owner_login=None if include_all else login, owner_login=None if include_all else login,
owner_email=None if include_all else email, owner_email=None if include_all else email,
), ),
_sidebar_active_thread_summary(
client,
active_thread_id,
fallback_threads={**active, **resolved_threads},
visible_thread_ids=active_ids,
login=login,
email=email,
include_all=include_all,
),
) )
active_has_more = len(active_candidates) > safe_active_limit
resolved_has_more = len(resolved_candidates) > safe_resolved_limit
if active_thread:
active_thread_summary, is_resolved_active_thread = active_thread
if is_resolved_active_thread:
resolved_items = [
active_thread_summary,
*[item for item in resolved_items if item["id"] != active_thread_summary["id"]],
]
active_items = [
item for item in active_items if item["id"] != active_thread_summary["id"]
]
if len(resolved_items) > safe_resolved_limit:
resolved_items = resolved_items[:safe_resolved_limit]
resolved_has_more = True
else:
active_items = [
active_thread_summary,
*[item for item in active_items if item["id"] != active_thread_summary["id"]],
]
resolved_items = [
item for item in resolved_items if item["id"] != active_thread_summary["id"]
]
if len(active_items) > safe_active_limit:
active_items = active_items[:safe_active_limit]
active_has_more = True
return { return {
"active": { "active": {
"items": active_items, "items": active_items,
"limit": safe_active_limit, "limit": safe_active_limit,
"hasMore": len(active_candidates) > safe_active_limit, "hasMore": active_has_more,
}, },
"resolved": { "resolved": {
"items": resolved_items, "items": resolved_items,
"limit": safe_resolved_limit, "limit": safe_resolved_limit,
"hasMore": len(resolved_candidates) > safe_resolved_limit, "hasMore": resolved_has_more,
}, },
} }

View file

@ -3,6 +3,7 @@
from __future__ import annotations from __future__ import annotations
import json import json
import os
import re import re
import shlex import shlex
from collections.abc import Awaitable, Callable, Mapping from collections.abc import Awaitable, Callable, Mapping
@ -14,6 +15,9 @@ from langgraph.prebuilt.tool_node import ToolCallRequest
from langgraph.types import Command from langgraph.types import Command
_SHELL_SEPARATORS = {";", "&&", "||", "|", "&"} _SHELL_SEPARATORS = {";", "&&", "||", "|", "&"}
_SHELL_EXECUTABLES = {"bash", "dash", "sh", "zsh"}
_MAX_SHELL_EXPANSION_DEPTH = 3
_SHELL_EXPANSION_DEPTH_LIMIT_TOKEN = "__pr_creation_guard_shell_expansion_depth_limit__"
_GITHUB_PULLS_ENDPOINT = re.compile(r"(?:^|/)repos/[^/\s]+/[^/\s]+/pulls/?$") _GITHUB_PULLS_ENDPOINT = re.compile(r"(?:^|/)repos/[^/\s]+/[^/\s]+/pulls/?$")
_GITHUB_PULLS_URL = re.compile(r"https://api\.github\.com/repos/[^/\s]+/[^/\s]+/pulls/?") _GITHUB_PULLS_URL = re.compile(r"https://api\.github\.com/repos/[^/\s]+/[^/\s]+/pulls/?")
_BLOCK_ERROR = ( _BLOCK_ERROR = (
@ -46,13 +50,62 @@ def _tool_call_id(request: ToolCallRequest) -> str | None:
return None return None
def _shell_tokens(command: str) -> list[str]: def _split_shell_tokens(command: str) -> list[str]:
try: try:
return shlex.split(command, posix=True) return shlex.split(command, posix=True)
except ValueError: except ValueError:
return command.split() return command.split()
def _executable_name(token: str) -> str:
return os.path.basename(token.strip("'\""))
def _shell_command_argument(tokens: list[str], shell_index: int) -> str | None:
for index, token in enumerate(tokens[shell_index + 1 :], start=shell_index + 1):
if token in _SHELL_SEPARATORS:
return None
if token == "-c" or (
token.startswith("-") and not token.startswith("--") and "c" in token[1:]
):
_, _, glued = token[1:].partition("c")
if glued:
return glued
if index + 1 < len(tokens) and tokens[index + 1] not in _SHELL_SEPARATORS:
return tokens[index + 1]
return None
return None
def _has_nested_shell_command(tokens: list[str]) -> bool:
return any(
_executable_name(token) in _SHELL_EXECUTABLES
and _shell_command_argument(tokens, index) is not None
for index, token in enumerate(tokens)
)
def _expand_nested_shell_tokens(tokens: list[str], depth: int = 0) -> list[str]:
if depth >= _MAX_SHELL_EXPANSION_DEPTH:
if _has_nested_shell_command(tokens):
return [*tokens, _SHELL_EXPANSION_DEPTH_LIMIT_TOKEN]
return tokens
expanded = list(tokens)
for index, token in enumerate(tokens):
if _executable_name(token) not in _SHELL_EXECUTABLES:
continue
inner_command = _shell_command_argument(tokens, index)
if inner_command is None:
continue
expanded.extend(_expand_nested_shell_tokens(_split_shell_tokens(inner_command), depth + 1))
return expanded
def _shell_tokens(command: str) -> list[str]:
return _expand_nested_shell_tokens(_split_shell_tokens(command))
def _is_assignment(token: str) -> bool: def _is_assignment(token: str) -> bool:
name, sep, _value = token.partition("=") name, sep, _value = token.partition("=")
return bool(sep and name and re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", name)) return bool(sep and name and re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", name))
@ -69,7 +122,7 @@ def _gh_subtokens(tokens: list[str], index: int) -> list[str]:
def _contains_gh_pr_create(tokens: list[str]) -> bool: def _contains_gh_pr_create(tokens: list[str]) -> bool:
for index, token in enumerate(tokens): for index, token in enumerate(tokens):
if token != "gh": if _executable_name(token) != "gh":
continue continue
subtokens = _gh_subtokens(tokens, index) subtokens = _gh_subtokens(tokens, index)
for offset, subtoken in enumerate(subtokens[:-1]): for offset, subtoken in enumerate(subtokens[:-1]):
@ -136,7 +189,7 @@ def _gh_api_uses_post_or_body(subtokens: list[str]) -> bool:
def _contains_gh_api_pull_create(tokens: list[str]) -> bool: def _contains_gh_api_pull_create(tokens: list[str]) -> bool:
for index, token in enumerate(tokens): for index, token in enumerate(tokens):
if token != "gh": if _executable_name(token) != "gh":
continue continue
subtokens = _gh_subtokens(tokens, index) subtokens = _gh_subtokens(tokens, index)
if "api" not in subtokens: if "api" not in subtokens:
@ -153,7 +206,7 @@ def _contains_gh_api_pull_create(tokens: list[str]) -> bool:
def _contains_direct_pull_create(tokens: list[str]) -> bool: def _contains_direct_pull_create(tokens: list[str]) -> bool:
for index, token in enumerate(tokens): for index, token in enumerate(tokens):
if token != "curl": if _executable_name(token) != "curl":
continue continue
subtokens: list[str] = [] subtokens: list[str] = []
for candidate in tokens[index + 1 :]: for candidate in tokens[index + 1 :]:
@ -193,7 +246,8 @@ def is_pr_creation_fallback_command(command: str) -> bool:
""" """
tokens = _shell_tokens(command) tokens = _shell_tokens(command)
return ( return (
_contains_gh_pr_create(tokens) _SHELL_EXPANSION_DEPTH_LIMIT_TOKEN in tokens
or _contains_gh_pr_create(tokens)
or _contains_gh_api_pull_create(tokens) or _contains_gh_api_pull_create(tokens)
or _contains_direct_pull_create(tokens) or _contains_direct_pull_create(tokens)
) )

View file

@ -3,6 +3,7 @@
from __future__ import annotations from __future__ import annotations
import json import json
import os
import re import re
import shlex import shlex
from collections.abc import Awaitable, Callable, Mapping from collections.abc import Awaitable, Callable, Mapping
@ -14,6 +15,9 @@ from langgraph.prebuilt.tool_node import ToolCallRequest
from langgraph.types import Command from langgraph.types import Command
_SHELL_SEPARATORS = {";", "&&", "||", "|", "&"} _SHELL_SEPARATORS = {";", "&&", "||", "|", "&"}
_SHELL_EXECUTABLES = {"bash", "dash", "sh", "zsh"}
_MAX_SHELL_EXPANSION_DEPTH = 3
_SHELL_EXPANSION_DEPTH_LIMIT_TOKEN = "__pr_verdict_guard_shell_expansion_depth_limit__"
_GITHUB_REVIEWS_ENDPOINT = re.compile( _GITHUB_REVIEWS_ENDPOINT = re.compile(
r"(?:^|/)repos/[^/\s]+/[^/\s]+/pulls/\d+/reviews(?:/\d+/events)?/?$" r"(?:^|/)repos/[^/\s]+/[^/\s]+/pulls/\d+/reviews(?:/\d+/events)?/?$"
) )
@ -55,13 +59,62 @@ def _tool_call_id(request: ToolCallRequest) -> str | None:
return None return None
def _shell_tokens(command: str) -> list[str]: def _split_shell_tokens(command: str) -> list[str]:
try: try:
return shlex.split(command, posix=True) return shlex.split(command, posix=True)
except ValueError: except ValueError:
return command.split() return command.split()
def _executable_name(token: str) -> str:
return os.path.basename(token.strip("'\""))
def _shell_command_argument(tokens: list[str], shell_index: int) -> str | None:
for index, token in enumerate(tokens[shell_index + 1 :], start=shell_index + 1):
if token in _SHELL_SEPARATORS:
return None
if token == "-c" or (
token.startswith("-") and not token.startswith("--") and "c" in token[1:]
):
_, _, glued = token[1:].partition("c")
if glued:
return glued
if index + 1 < len(tokens) and tokens[index + 1] not in _SHELL_SEPARATORS:
return tokens[index + 1]
return None
return None
def _has_nested_shell_command(tokens: list[str]) -> bool:
return any(
_executable_name(token) in _SHELL_EXECUTABLES
and _shell_command_argument(tokens, index) is not None
for index, token in enumerate(tokens)
)
def _expand_nested_shell_tokens(tokens: list[str], depth: int = 0) -> list[str]:
if depth >= _MAX_SHELL_EXPANSION_DEPTH:
if _has_nested_shell_command(tokens):
return [*tokens, _SHELL_EXPANSION_DEPTH_LIMIT_TOKEN]
return tokens
expanded = list(tokens)
for index, token in enumerate(tokens):
if _executable_name(token) not in _SHELL_EXECUTABLES:
continue
inner_command = _shell_command_argument(tokens, index)
if inner_command is None:
continue
expanded.extend(_expand_nested_shell_tokens(_split_shell_tokens(inner_command), depth + 1))
return expanded
def _shell_tokens(command: str) -> list[str]:
return _expand_nested_shell_tokens(_split_shell_tokens(command))
def _subtokens_after(tokens: list[str], index: int) -> list[str]: def _subtokens_after(tokens: list[str], index: int) -> list[str]:
subtokens: list[str] = [] subtokens: list[str] = []
for token in tokens[index + 1 :]: for token in tokens[index + 1 :]:
@ -73,7 +126,7 @@ def _subtokens_after(tokens: list[str], index: int) -> list[str]:
def _contains_gh_pr_review_verdict(tokens: list[str]) -> bool: def _contains_gh_pr_review_verdict(tokens: list[str]) -> bool:
for index, token in enumerate(tokens): for index, token in enumerate(tokens):
if token != "gh": if _executable_name(token) != "gh":
continue continue
subtokens = _subtokens_after(tokens, index) subtokens = _subtokens_after(tokens, index)
is_pr_review = any( is_pr_review = any(
@ -92,7 +145,7 @@ def _contains_gh_pr_review_verdict(tokens: list[str]) -> bool:
def _contains_gh_api_review_verdict(tokens: list[str]) -> bool: def _contains_gh_api_review_verdict(tokens: list[str]) -> bool:
for index, token in enumerate(tokens): for index, token in enumerate(tokens):
if token != "gh": if _executable_name(token) != "gh":
continue continue
subtokens = _subtokens_after(tokens, index) subtokens = _subtokens_after(tokens, index)
if "api" not in subtokens: if "api" not in subtokens:
@ -111,7 +164,7 @@ def _contains_gh_api_review_verdict(tokens: list[str]) -> bool:
def _contains_direct_review_verdict(tokens: list[str]) -> bool: def _contains_direct_review_verdict(tokens: list[str]) -> bool:
for index, token in enumerate(tokens): for index, token in enumerate(tokens):
if token != "curl": if _executable_name(token) != "curl":
continue continue
subtokens = _subtokens_after(tokens, index) subtokens = _subtokens_after(tokens, index)
if not any(_GITHUB_REVIEWS_URL.search(subtoken) for subtoken in subtokens): if not any(_GITHUB_REVIEWS_URL.search(subtoken) for subtoken in subtokens):
@ -129,7 +182,13 @@ def is_pr_verdict_fallback_command(command: str) -> bool:
``event=APPROVE|REQUEST_CHANGES`` to a ``/pulls/N/reviews`` endpoint, and ``event=APPROVE|REQUEST_CHANGES`` to a ``/pulls/N/reviews`` endpoint, and
``curl`` to the reviews URL with a verdict event in the body) but is ``curl`` to the reviews URL with a verdict event in the body) but is
intentionally fail-open: shell aliases, ``gh`` aliases, ``--input`` intentionally fail-open: shell aliases, ``gh`` aliases, ``--input``
JSON-file bodies, and non-curl HTTP clients will not be blocked. This is JSON-file bodies, and non-curl HTTP clients will not be blocked. The
``-c`` form of a known shell (``bash``/``dash``/``sh``/``zsh``, whether
space-separated or glued as ``-c'...'``) is expanded to a fixed depth and
quoted/path-prefixed executables are normalized, so those do not slip
through; other shells (``ash``/``busybox``/``fish``) and stdin-fed
programs (``... | bash``, here-strings, ``bash -s``) remain fail-open.
This is
acceptable because the threat model is an honest agent papering over a acceptable because the threat model is an honest agent papering over a
``publish_review`` limitation or refusal, not an adversary trying to ``publish_review`` limitation or refusal, not an adversary trying to
bypass the guardrail. Plain ``gh pr review`` (no verdict flag) and bypass the guardrail. Plain ``gh pr review`` (no verdict flag) and
@ -138,7 +197,8 @@ def is_pr_verdict_fallback_command(command: str) -> bool:
""" """
tokens = _shell_tokens(command) tokens = _shell_tokens(command)
return ( return (
_contains_gh_pr_review_verdict(tokens) _SHELL_EXPANSION_DEPTH_LIMIT_TOKEN in tokens
or _contains_gh_pr_review_verdict(tokens)
or _contains_gh_api_review_verdict(tokens) or _contains_gh_api_review_verdict(tokens)
or _contains_direct_review_verdict(tokens) or _contains_direct_review_verdict(tokens)
) )

View file

@ -43,7 +43,10 @@ async def save_plan(
shared content. shared content.
Write the content in standard Markdown — headings, bullet/numbered lists, and Write the content in standard Markdown — headings, bullet/numbered lists, and
fenced code blocks all render. fenced code blocks all render. Shared responses persist Markdown text only;
they do not upload or serve sandbox-local or relative image files. If images
or screenshots are needed for a Slack response, post them directly in Slack
instead of relying on the shared response page.
Args: Args:
plan_file_path: Path to the Markdown plan file in the sandbox. plan_file_path: Path to the Markdown plan file in the sandbox.

View file

@ -2,9 +2,11 @@ import os
import re import re
from typing import Any from typing import Any
from fastapi import HTTPException
from langgraph.config import get_config from langgraph.config import get_config
from langgraph_sdk import get_client from langgraph_sdk import get_client
from ..dashboard.repo_access import require_repo_access_for_user
from ..dispatch import dispatch_agent_run from ..dispatch import dispatch_agent_run
from ..utils.dashboard_links import dashboard_thread_url from ..utils.dashboard_links import dashboard_thread_url
from ..utils.slack import ( from ..utils.slack import (
@ -13,6 +15,7 @@ from ..utils.slack import (
store_slack_run_mapping, store_slack_run_mapping,
) )
from ..utils.thread_ids import generate_thread_id_from_slack_thread from ..utils.thread_ids import generate_thread_id_from_slack_thread
from ..webhooks.common import _is_repo_allowed
LANGGRAPH_URL = os.environ.get("LANGGRAPH_URL") or os.environ.get( LANGGRAPH_URL = os.environ.get("LANGGRAPH_URL") or os.environ.get(
"LANGGRAPH_URL_PROD", "http://localhost:2024" "LANGGRAPH_URL_PROD", "http://localhost:2024"
@ -160,6 +163,42 @@ async def slack_start_new_thread(
"error": "default_repo must be a simple owner/name repository string", "error": "default_repo must be a simple owner/name repository string",
} }
if default_repo and default_repo.strip() and repo is not None:
if not _is_repo_allowed(repo):
return {
"success": False,
"error": (
f"Repository {repo['owner']}/{repo['name']} is not on the deployment allowlist"
),
}
github_login = configurable.get("github_login")
if not isinstance(github_login, str) or not github_login.strip():
return {
"success": False,
"error": (
"Cannot verify access to the requested repository: no github_login on the "
"parent thread"
),
}
try:
await require_repo_access_for_user(
github_login.strip(), f"{repo['owner']}/{repo['name']}"
)
except HTTPException as exc:
return {
"success": False,
"error": (
f"Access to repository {repo['owner']}/{repo['name']} denied: {exc.detail}"
),
}
except Exception as exc: # noqa: BLE001
return {
"success": False,
"error": (
f"Failed to verify access to repository {repo['owner']}/{repo['name']}: {exc}"
),
}
message_ts, slack_error = await post_slack_top_level_message_with_ts( message_ts, slack_error = await post_slack_top_level_message_with_ts(
channel_id.strip(), channel_id.strip(),
_visible_message(clean_title, clean_instructions, repo), _visible_message(clean_title, clean_instructions, repo),

View file

@ -589,7 +589,9 @@ async def process_github_push_event(payload: dict[str, Any]) -> None:
await common.reconcile_findings_with_review_threads(thread_id, threads) await common.reconcile_findings_with_review_threads(thread_id, threads)
except Exception: except Exception:
common.logger.warning( common.logger.warning(
"Could not sync review threads before push re-review for %s", thread_id "Could not sync review threads before push re-review for %s",
thread_id,
exc_info=True,
) )
pr_meta: ReviewerPRMeta = { pr_meta: ReviewerPRMeta = {

View file

@ -130,14 +130,14 @@
{"sha": "f0897479", "pr": 1778, "subject": "feat: surface context window usage in agents UI (#1778)", "disposition": "deferred", "reason": "Near-clean: context-window indicator UI is all new files (zero drift); options.py hunk must be re-keyed to the fork's Bedrock/Fireworks model map (fork model IDs differ from upstream's) — add context_window per fork entry rather than taking upstream values.", "branch": "context-usage-ui", "local_sha": null, "updated": "2026-07-17T22:33:46Z"} {"sha": "f0897479", "pr": 1778, "subject": "feat: surface context window usage in agents UI (#1778)", "disposition": "deferred", "reason": "Near-clean: context-window indicator UI is all new files (zero drift); options.py hunk must be re-keyed to the fork's Bedrock/Fireworks model map (fork model IDs differ from upstream's) — add context_window per fork entry rather than taking upstream values.", "branch": "context-usage-ui", "local_sha": null, "updated": "2026-07-17T22:33:46Z"}
{"sha": "31263f83", "pr": 1785, "subject": "Fix: Fix Stored XSS in ReplyCard.tsx (#1785)", "disposition": "landed", "reason": "SECURITY priority, near-clean: diff is urlencode() hardening of the GitHub OAuth authorize URL in dashboard routes.py auth_login (upstream bot title says ReplyCard.tsx — mismatch, trust the diff). Fork auth_login has the identical f-string URL; hunk applies clean. Port promptly.", "branch": "feature/upstream-clean-batch-security-linear", "local_sha": null, "updated": "2026-07-20T19:46:00Z"} {"sha": "31263f83", "pr": 1785, "subject": "Fix: Fix Stored XSS in ReplyCard.tsx (#1785)", "disposition": "landed", "reason": "SECURITY priority, near-clean: diff is urlencode() hardening of the GitHub OAuth authorize URL in dashboard routes.py auth_login (upstream bot title says ReplyCard.tsx — mismatch, trust the diff). Fork auth_login has the identical f-string URL; hunk applies clean. Port promptly.", "branch": "feature/upstream-clean-batch-security-linear", "local_sha": null, "updated": "2026-07-20T19:46:00Z"}
{"sha": "3ea29d3f", "pr": 1789, "subject": "Fix: Fix Improper privilege management in server.py (#1789)", "disposition": "landed", "reason": "SECURITY priority, near-clean: adds empty-signature reject to verify_github_signature (utils/github_comments.py) + verify_linear_signature (webhooks/common.py) (title says server.py — mismatch, trust the diff). Both fork functions match at the hunk sites; fork-only verify_jira_secret already guards empty token. Real value: None signature currently raises TypeError in compare_digest. Port BOTH hunks together.", "branch": "feature/upstream-clean-batch-security-linear", "local_sha": null, "updated": "2026-07-20T19:46:00Z"} {"sha": "3ea29d3f", "pr": 1789, "subject": "Fix: Fix Improper privilege management in server.py (#1789)", "disposition": "landed", "reason": "SECURITY priority, near-clean: adds empty-signature reject to verify_github_signature (utils/github_comments.py) + verify_linear_signature (webhooks/common.py) (title says server.py — mismatch, trust the diff). Both fork functions match at the hunk sites; fork-only verify_jira_secret already guards empty token. Real value: None signature currently raises TypeError in compare_digest. Port BOTH hunks together.", "branch": "feature/upstream-clean-batch-security-linear", "local_sha": null, "updated": "2026-07-20T19:46:00Z"}
{"sha": "2e8ff4b7", "pr": 1782, "subject": "chore: clarify shared response image guidance (#1782)", "disposition": "deferred", "reason": "Near-clean: docstring-only guidance in save_plan (shared responses persist Markdown only, never sandbox-local images — post screenshots directly to Slack) + 2 assertions in test_plan_review. Merges clean against fork. Batch with the next clean-port round.", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-21T18:00:13Z"} {"sha": "2e8ff4b7", "pr": 1782, "subject": "chore: clarify shared response image guidance (#1782)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"}
{"sha": "4ea2441a", "pr": 1791, "subject": "fix: match embedded review description background (#1791)", "disposition": "deferred", "reason": "Near-clean UI-only: ReviewMainBody.tsx background match for the embedded review description; merges clean, zero fork drift. Batch with the next clean-port round.", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-21T18:00:13Z"} {"sha": "4ea2441a", "pr": 1791, "subject": "fix: match embedded review description background (#1791)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"}
{"sha": "589dfd83", "pr": 1787, "subject": "fix: Harden Stagehand browser URL handling (#1787)", "disposition": "wont-merge", "reason": "Follows #1648 (wont-merge): fork does not carry the Stagehand browser subagent — this 'fix' re-adds the entire module (992 insertions; modify/delete vs HEAD) plus the stagehand dep in pyproject/uv.lock. Adopting would resurrect a feature rejected under the curated-tools policy.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:14Z"} {"sha": "589dfd83", "pr": 1787, "subject": "fix: Harden Stagehand browser URL handling (#1787)", "disposition": "wont-merge", "reason": "Follows #1648 (wont-merge): fork does not carry the Stagehand browser subagent — this 'fix' re-adds the entire module (992 insertions; modify/delete vs HEAD) plus the stagehand dep in pyproject/uv.lock. Adopting would resurrect a feature rejected under the curated-tools policy.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:14Z"}
{"sha": "9bbd65d3", "pr": 1781, "subject": "fix: derive model context windows from LangChain profiles (#1781)", "disposition": "deferred", "reason": "Pair with deferred #1778 on context-usage-ui: derives model context windows from LangChain profiles. options.py conflicts — re-key to the fork's Bedrock/Fireworks model map (upstream IDs differ; verify LangChain profiles even resolve for the fork's Bedrock-style IDs, else keep static values). uv.lock bumps langchain to a profiles-capable version; test_model_fallback_resolution also conflicts. Port with/after #1778.", "branch": "context-usage-ui", "local_sha": null, "updated": "2026-07-21T18:00:14Z"} {"sha": "9bbd65d3", "pr": 1781, "subject": "fix: derive model context windows from LangChain profiles (#1781)", "disposition": "deferred", "reason": "Pair with deferred #1778 on context-usage-ui: derives model context windows from LangChain profiles. options.py conflicts — re-key to the fork's Bedrock/Fireworks model map (upstream IDs differ; verify LangChain profiles even resolve for the fork's Bedrock-style IDs, else keep static values). uv.lock bumps langchain to a profiles-capable version; test_model_fallback_resolution also conflicts. Port with/after #1778.", "branch": "context-usage-ui", "local_sha": null, "updated": "2026-07-21T18:00:14Z"}
{"sha": "df743658", "pr": 1774, "subject": "feat: add repository skill support to the coding agent (#1774)", "disposition": "wont-merge", "reason": "Upstream reverted this feature in #1800: its factory-time ensure_sandbox_for_thread call ran outside the serialized run and raced the deterministic sandbox name (prod sandbox-create 409s spiking from the #1774 merge date). Feature withdrawn upstream; the fork's deferred security review is moot. If upstream re-lands repo skills (host-side .agents/skills resolution per #1800), triage that new commit fresh — the prompt-injection / trusted-ref concerns in the old reason still apply.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"} {"sha": "df743658", "pr": 1774, "subject": "feat: add repository skill support to the coding agent (#1774)", "disposition": "wont-merge", "reason": "Upstream reverted this feature in #1800: its factory-time ensure_sandbox_for_thread call ran outside the serialized run and raced the deterministic sandbox name (prod sandbox-create 409s spiking from the #1774 merge date). Feature withdrawn upstream; the fork's deferred security review is moot. If upstream re-lands repo skills (host-side .agents/skills resolution per #1800), triage that new commit fresh — the prompt-injection / trusted-ref concerns in the old reason still apply.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"}
{"sha": "75fb8b48", "pr": 1786, "subject": "Fix PR creation guard shell bypasses (#1786)", "disposition": "deferred", "reason": "SECURITY priority, clean merge: closes shell-bypass holes in PullRequestCreationGuardMiddleware (nested 'bash -c' expansion to depth 3, quoted-executable normalization) — a guard this fork actively wires. Port promptly; ALSO mirror the nested-shell expansion into the fork-only PullRequestVerdictGuardMiddleware (pr_verdict_guard.py), which shares the naive shlex approach and has the same bypass shape.", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-21T18:00:23Z"} {"sha": "75fb8b48", "pr": 1786, "subject": "Fix PR creation guard shell bypasses (#1786)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"}
{"sha": "32e81f29", "pr": 1788, "subject": "Fix: Fix Insecure Direct Object Reference in slack_start_new_thread.py (#1788)", "disposition": "deferred", "reason": "SECURITY priority, clean merge: IDOR fix — slack_start_new_thread now enforces the deployment allowlist (_is_repo_allowed) and per-user repo access (require_repo_access_for_user keyed on the parent thread's github_login) before dispatching a run against a different repo. Both helpers exist in the fork at the same paths; merges clean. Port promptly.", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-21T18:00:23Z"} {"sha": "32e81f29", "pr": 1788, "subject": "Fix: Fix Insecure Direct Object Reference in slack_start_new_thread.py (#1788)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"}
{"sha": "ab85b372", "pr": 1764, "subject": "fix: add exc_info to swallowed exception in push re-review webhook (#1764)", "disposition": "deferred", "reason": "Trivial: adds exc_info=True to the swallowed exception log in process_github_push_event (webhooks/github.py). Merges clean. Batch with the next clean-port round.", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-21T18:00:23Z"} {"sha": "ab85b372", "pr": 1764, "subject": "fix: add exc_info to swallowed exception in push re-review webhook (#1764)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"}
{"sha": "7312851d", "pr": 1777, "subject": "feat: add explicit plan approval tool (#1777)", "disposition": "deferred", "reason": "Feature: explicit approve_plan tool + plan-mode exit. All deps exist in fork (plan_store PLAN_STATUS_APPROVED/SHARED, thread_api._user_owns_thread). Conflicts: prompt.py (fork prompt constants), server.py (tool/middleware wiring), check_message_queue.py (fork dashboard-handoff + mid-run injection drift), AGENTS.md. Hand re-key those four; keep the whole vertical (tool + plan_mode + thread_api + 4 test files) on one side.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:33Z"} {"sha": "7312851d", "pr": 1777, "subject": "feat: add explicit plan approval tool (#1777)", "disposition": "deferred", "reason": "Feature: explicit approve_plan tool + plan-mode exit. All deps exist in fork (plan_store PLAN_STATUS_APPROVED/SHARED, thread_api._user_owns_thread). Conflicts: prompt.py (fork prompt constants), server.py (tool/middleware wiring), check_message_queue.py (fork dashboard-handoff + mid-run injection drift), AGENTS.md. Hand re-key those four; keep the whole vertical (tool + plan_mode + thread_api + 4 test files) on one side.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:33Z"}
{"sha": "e51abe14", "pr": 1754, "subject": "fix: skip oversized images before model calls (#1754)", "disposition": "deferred", "reason": "Desirable: 10MB image size cap + 'image omitted' text block before model calls, prevents oversized-image model failures. Conflicts only because the fork hardened fetch_image_block (SSRF redirect guard, host-only logging) — hunks are compatible; re-key the size check around the fork's guarded fetch and port impl + test together.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:33Z"} {"sha": "e51abe14", "pr": 1754, "subject": "fix: skip oversized images before model calls (#1754)", "disposition": "deferred", "reason": "Desirable: 10MB image size cap + 'image omitted' text block before model calls, prevents oversized-image model failures. Conflicts only because the fork hardened fetch_image_block (SSRF redirect guard, host-only logging) — hunks are compatible; re-key the size check around the fork's guarded fetch and port impl + test together.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:33Z"}
{"sha": "0ed560a5", "pr": 1779, "subject": "fix: settle review checks after run failures (#1779)", "disposition": "deferred", "reason": "Desirable: run-failure completion webhook now settles reviewer check runs left open when the graph dies (_settle_failed_reviewer_check; uses settle_review_check_run + review_check_pending_result, both already in the fork's review/publish.py from the #214/#215 ports). Conflict is fork drift in completion.py (failure-reply customizations); re-key the new helper in and port with its 159-line test.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:33Z"} {"sha": "0ed560a5", "pr": 1779, "subject": "fix: settle review checks after run failures (#1779)", "disposition": "deferred", "reason": "Desirable: run-failure completion webhook now settles reviewer check runs left open when the graph dies (_settle_failed_reviewer_check; uses settle_review_check_run + review_check_pending_result, both already in the fork's review/publish.py from the #214/#215 ports). Conflict is fork drift in completion.py (failure-reply customizations); re-key the new helper in and port with its 159-line test.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:33Z"}
@ -146,7 +146,7 @@
{"sha": "bedc57b8", "pr": 1796, "subject": "fix: cap execute output by making SandboxBackendProxy a BaseSandbox (#1796)", "disposition": "deferred", "reason": "Real fix (unbounded execute stdout pulled into the worker → OOM risk) but inapplicable at the fork's deepagents==0.6.12: no capture-offload API exists (ExecuteOffloadResult / execute_accepts_timeout absent; FilesystemMiddleware has no _resolve_capture BaseSandbox gate), so the bug it fixes cannot occur yet. Re-triage together with deferred #1745 (deepagents 0.6.12→0.7.x bump) — landing that bump makes this fix required. Fork's SandboxBackendProxy has diverged (sync-era, bound_repo repo-binding, no reconnect machinery): hand-apply the execute_with_offload/aexecute_with_offload delegation onto the fork proxy rather than cherry-pick.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"} {"sha": "bedc57b8", "pr": 1796, "subject": "fix: cap execute output by making SandboxBackendProxy a BaseSandbox (#1796)", "disposition": "deferred", "reason": "Real fix (unbounded execute stdout pulled into the worker → OOM risk) but inapplicable at the fork's deepagents==0.6.12: no capture-offload API exists (ExecuteOffloadResult / execute_accepts_timeout absent; FilesystemMiddleware has no _resolve_capture BaseSandbox gate), so the bug it fixes cannot occur yet. Re-triage together with deferred #1745 (deepagents 0.6.12→0.7.x bump) — landing that bump makes this fix required. Fork's SandboxBackendProxy has diverged (sync-era, bound_repo repo-binding, no reconnect machinery): hand-apply the execute_with_offload/aexecute_with_offload delegation onto the fork proxy rather than cherry-pick.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"}
{"sha": "e1138cf5", "pr": 1797, "subject": "fix: merge concurrent trusted skill refs (#1797)", "disposition": "wont-merge", "reason": "Follow-up fix to #1774's TrustedSkillsMiddleware, which the fork never adopted (row df743658) and upstream itself reverted in #1800 — nothing to apply.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"} {"sha": "e1138cf5", "pr": 1797, "subject": "fix: merge concurrent trusted skill refs (#1797)", "disposition": "wont-merge", "reason": "Follow-up fix to #1774's TrustedSkillsMiddleware, which the fork never adopted (row df743658) and upstream itself reverted in #1800 — nothing to apply.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"}
{"sha": "60e7307c", "pr": 1800, "subject": "revert: repository skill support for the coding agent (#1774, #1797) (#1800)", "disposition": "wont-merge", "reason": "Revert of #1774 + #1797; the fork adopted neither, so there is nothing to revert. Upstream's rationale: #1774's factory-time ensure_sandbox_for_thread ran outside the serialized run and raced the thread-deterministic sandbox name (prod create-409s). Fork invariant holds: its single provisioning call sits inside the __creating__-sentinel lifecycle within the interrupt-serialized run.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"} {"sha": "60e7307c", "pr": 1800, "subject": "revert: repository skill support for the coding agent (#1774, #1797) (#1800)", "disposition": "wont-merge", "reason": "Revert of #1774 + #1797; the fork adopted neither, so there is nothing to revert. Upstream's rationale: #1774's factory-time ensure_sandbox_for_thread ran outside the serialized run and raced the thread-deterministic sandbox name (prod create-409s). Fork invariant holds: its single provisioning call sits inside the __creating__-sentinel lifecycle within the interrupt-serialized run.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"}
{"sha": "a77c4e47", "pr": 1799, "subject": "fix: show current shared thread in sidebar (#1799)", "disposition": "deferred", "reason": "Desirable dashboard fix: sidebar now surfaces the currently-open shared thread (org-member 'Open in Web' links) when it falls outside the caller's own list — _sidebar_active_thread_summary + active_thread_id param; fork has the same include_all/org-member read path in thread_api.py. Cherry-picks CLEAN onto dev (merge-tree probe: zero conflicts); port with its thread-api tests + sandbox_id e2e spec + AgentsSidebar/queries UI as one vertical.", "branch": "", "local_sha": null, "updated": "2026-07-24T19:25:32Z"} {"sha": "a77c4e47", "pr": 1799, "subject": "fix: show current shared thread in sidebar (#1799)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"}
{"sha": "b67215ed", "pr": 1780, "subject": "feat: link originating Slack threads (#1780)", "disposition": "deferred", "reason": "Desirable for the Slack-heavy fork: dispatch records the originating Slack permalink in thread metadata (webhooks/common.py), open_pull_request embeds an 'Originating Slack thread' link in the PR body (gated to private repos), and the sidebar menu links it. Impl files auto-merge (thread_api / open_pull_request / webhooks/common); conflicts are fork drift in prompt.py (customized prompt constants), tests/e2e/harness.py, tests/slack/test_slack_context.py — hand re-key those; keep backend + UI (AgentsSidebar, types) + e2e dashboard spec as one vertical.", "branch": "", "local_sha": null, "updated": "2026-07-24T19:25:37Z"} {"sha": "b67215ed", "pr": 1780, "subject": "feat: link originating Slack threads (#1780)", "disposition": "deferred", "reason": "Desirable for the Slack-heavy fork: dispatch records the originating Slack permalink in thread metadata (webhooks/common.py), open_pull_request embeds an 'Originating Slack thread' link in the PR body (gated to private repos), and the sidebar menu links it. Impl files auto-merge (thread_api / open_pull_request / webhooks/common); conflicts are fork drift in prompt.py (customized prompt constants), tests/e2e/harness.py, tests/slack/test_slack_context.py — hand re-key those; keep backend + UI (AgentsSidebar, types) + e2e dashboard spec as one vertical.", "branch": "", "local_sha": null, "updated": "2026-07-24T19:25:37Z"}
{"sha": "7d79d300", "pr": 1804, "subject": "chore(deps): bump deepagents to 0.7.0b1 pin (#1804)", "disposition": "deferred", "reason": "Rides the deferred #1745 deepagents bump: 0.7.0a7→0.7.0b1 pin + wcmatch>=11.0 uv override (langchain-e2b's e2b dep still caps wcmatch<11.0). Fork is pinned deepagents==0.6.12 — and langsmith==0.10.10 already satisfies the new >=0.10.9 floor — and never took the a7-era HARNESS_EXCLUDED_MIDDLEWARE code its prompt.py hunk deletes, so nothing applies standalone. Re-triage as one cluster: #1745 (bump, now targeting 0.7.0b1) + #1796 (offload delegation) + this pin/override.", "branch": "", "local_sha": null, "updated": "2026-07-24T19:25:44Z"} {"sha": "7d79d300", "pr": 1804, "subject": "chore(deps): bump deepagents to 0.7.0b1 pin (#1804)", "disposition": "deferred", "reason": "Rides the deferred #1745 deepagents bump: 0.7.0a7→0.7.0b1 pin + wcmatch>=11.0 uv override (langchain-e2b's e2b dep still caps wcmatch<11.0). Fork is pinned deepagents==0.6.12 — and langsmith==0.10.10 already satisfies the new >=0.10.9 floor — and never took the a7-era HARNESS_EXCLUDED_MIDDLEWARE code its prompt.py hunk deletes, so nothing applies standalone. Re-triage as one cluster: #1745 (bump, now targeting 0.7.0b1) + #1796 (offload delegation) + this pin/override.", "branch": "", "local_sha": null, "updated": "2026-07-24T19:25:44Z"}
{"sha": "0e2eefbc", "pr": 1801, "subject": "chore: bump default sandbox resources and delete-after-stop (#1801)", "disposition": "deferred", "reason": "Capacity/cost policy, not a correctness fix: defaults 2→4 vCPU, 7936MiB→16GiB mem, 32→128GiB FS, stopped-sandbox delete 24h→14d. Would materially raise LangSmith sandbox spend on sh-openswe unless prod env already overrides; all five values are env-overridable today. Decide prod sizing first (ops decision), then adopt or reject. Textual conflict is only fork drift in integrations/langsmith.py (sync-era SandboxClient, reconnect constants absent) — the constants re-key trivially; docs hunks auto-merge.", "branch": "", "local_sha": null, "updated": "2026-07-24T19:25:49Z"} {"sha": "0e2eefbc", "pr": 1801, "subject": "chore: bump default sandbox resources and delete-after-stop (#1801)", "disposition": "deferred", "reason": "Capacity/cost policy, not a correctness fix: defaults 2→4 vCPU, 7936MiB→16GiB mem, 32→128GiB FS, stopped-sandbox delete 24h→14d. Would materially raise LangSmith sandbox spend on sh-openswe unless prod env already overrides; all five values are env-overridable today. Decide prod sizing first (ops decision), then adopt or reject. Textual conflict is only fork drift in integrations/langsmith.py (sync-era SandboxClient, reconnect constants absent) — the constants re-key trivially; docs hunks auto-merge.", "branch": "", "local_sha": null, "updated": "2026-07-24T19:25:49Z"}

View file

@ -88,6 +88,12 @@ Rows key on the **upstream SHA** (stable across local cherry-picks). Deferred ro
| `b5e52925` | #1775 | feat: add structured Linear issue filters (#1775) | Landed | Clean pick: structured filters for linear_search_issues — stacks directly on #1748 (landed PR #206); zero fork drift on all three files. Ready whenever. | feature/upstream-clean-batch-security-linear | | `b5e52925` | #1775 | feat: add structured Linear issue filters (#1775) | Landed | Clean pick: structured filters for linear_search_issues — stacks directly on #1748 (landed PR #206); zero fork drift on all three files. Ready whenever. | feature/upstream-clean-batch-security-linear |
| `31263f83` | #1785 | Fix: Fix Stored XSS in ReplyCard.tsx (#1785) | Landed | SECURITY priority, near-clean: diff is urlencode() hardening of the GitHub OAuth authorize URL in dashboard routes.py auth_login (upstream bot title says ReplyCard.tsx — mismatch, trust the diff). Fork auth_login has the identical f-string URL; hunk applies clean. Port promptly. | feature/upstream-clean-batch-security-linear | | `31263f83` | #1785 | Fix: Fix Stored XSS in ReplyCard.tsx (#1785) | Landed | SECURITY priority, near-clean: diff is urlencode() hardening of the GitHub OAuth authorize URL in dashboard routes.py auth_login (upstream bot title says ReplyCard.tsx — mismatch, trust the diff). Fork auth_login has the identical f-string URL; hunk applies clean. Port promptly. | feature/upstream-clean-batch-security-linear |
| `3ea29d3f` | #1789 | Fix: Fix Improper privilege management in server.py (#1789) | Landed | SECURITY priority, near-clean: adds empty-signature reject to verify_github_signature (utils/github_comments.py) + verify_linear_signature (webhooks/common.py) (title says server.py — mismatch, trust the diff). Both fork functions match at the hunk sites; fork-only verify_jira_secret already guards empty token. Real value: None signature currently raises TypeError in compare_digest. Port BOTH hunks together. | feature/upstream-clean-batch-security-linear | | `3ea29d3f` | #1789 | Fix: Fix Improper privilege management in server.py (#1789) | Landed | SECURITY priority, near-clean: adds empty-signature reject to verify_github_signature (utils/github_comments.py) + verify_linear_signature (webhooks/common.py) (title says server.py — mismatch, trust the diff). Both fork functions match at the hunk sites; fork-only verify_jira_secret already guards empty token. Real value: None signature currently raises TypeError in compare_digest. Port BOTH hunks together. | feature/upstream-clean-batch-security-linear |
| `2e8ff4b7` | #1782 | chore: clarify shared response image guidance (#1782) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 |
| `4ea2441a` | #1791 | fix: match embedded review description background (#1791) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 |
| `75fb8b48` | #1786 | Fix PR creation guard shell bypasses (#1786) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 |
| `32e81f29` | #1788 | Fix: Fix Insecure Direct Object Reference in slack_start_new_thread.py (#1788) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 |
| `ab85b372` | #1764 | fix: add exc_info to swallowed exception in push re-review webhook (#1764) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 |
| `a77c4e47` | #1799 | fix: show current shared thread in sidebar (#1799) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 |
| `c3292d82` | #1611 | bake sfw binary into sandbox image | Won't merge | already in dev | | | `c3292d82` | #1611 | bake sfw binary into sandbox image | Won't merge | already in dev | |
| `48bf712b` | #1609 | show message timestamps | Won't merge | already in dev | | | `48bf712b` | #1609 | show message timestamps | Won't merge | already in dev | |
| `85c0f63e` | #1620 | clickable shared PR header | Won't merge | already in dev | | | `85c0f63e` | #1620 | clickable shared PR header | Won't merge | already in dev | |
@ -147,17 +153,11 @@ Rows key on the **upstream SHA** (stable across local cherry-picks). Deferred ro
| `81d544bc` | #1765 | feat: Expose more specific AGENTS.md context to Open SWE reviewer (#1765) | Deferred | Near-clean: agents_md.py helper + tests are zero-drift; reviewer.py hunk is small (~39 lines) but lands in the fork's heavily-diverged reviewer — hand-apply the scoped_agents_md wiring onto the fork's fetch_agents_md call sites (reviewer.py ~L404/445/1031). | reviewer-context | | `81d544bc` | #1765 | feat: Expose more specific AGENTS.md context to Open SWE reviewer (#1765) | Deferred | Near-clean: agents_md.py helper + tests are zero-drift; reviewer.py hunk is small (~39 lines) but lands in the fork's heavily-diverged reviewer — hand-apply the scoped_agents_md wiring onto the fork's fetch_agents_md call sites (reviewer.py ~L404/445/1031). | reviewer-context |
| `8c8e58bc` | #1776 | feat: connect automations to Slack channels (#1776) | Deferred | Moderate reconcile: Slack-channel wiring for automations. Fork helpers exist (post_slack_top_level_message_with_ts, generate_thread_id_from_slack_thread, slack_id_for_login); completion.py (+233 drift) and schedules.py (+167) need hand-merge; UI automations feature present. Keep backend+UI+tests as one vertical. | automations-slack | | `8c8e58bc` | #1776 | feat: connect automations to Slack channels (#1776) | Deferred | Moderate reconcile: Slack-channel wiring for automations. Fork helpers exist (post_slack_top_level_message_with_ts, generate_thread_id_from_slack_thread, slack_id_for_login); completion.py (+233 drift) and schedules.py (+167) need hand-merge; UI automations feature present. Keep backend+UI+tests as one vertical. | automations-slack |
| `f0897479` | #1778 | feat: surface context window usage in agents UI (#1778) | Deferred | Near-clean: context-window indicator UI is all new files (zero drift); options.py hunk must be re-keyed to the fork's Bedrock/Fireworks model map (fork model IDs differ from upstream's) — add context_window per fork entry rather than taking upstream values. | context-usage-ui | | `f0897479` | #1778 | feat: surface context window usage in agents UI (#1778) | Deferred | Near-clean: context-window indicator UI is all new files (zero drift); options.py hunk must be re-keyed to the fork's Bedrock/Fireworks model map (fork model IDs differ from upstream's) — add context_window per fork entry rather than taking upstream values. | context-usage-ui |
| `2e8ff4b7` | #1782 | chore: clarify shared response image guidance (#1782) | Deferred | Near-clean: docstring-only guidance in save_plan (shared responses persist Markdown only, never sandbox-local images — post screenshots directly to Slack) + 2 assertions in test_plan_review. Merges clean against fork. Batch with the next clean-port round. | feature/upstream-clean-batch-jul21 |
| `4ea2441a` | #1791 | fix: match embedded review description background (#1791) | Deferred | Near-clean UI-only: ReviewMainBody.tsx background match for the embedded review description; merges clean, zero fork drift. Batch with the next clean-port round. | feature/upstream-clean-batch-jul21 |
| `9bbd65d3` | #1781 | fix: derive model context windows from LangChain profiles (#1781) | Deferred | Pair with deferred #1778 on context-usage-ui: derives model context windows from LangChain profiles. options.py conflicts — re-key to the fork's Bedrock/Fireworks model map (upstream IDs differ; verify LangChain profiles even resolve for the fork's Bedrock-style IDs, else keep static values). uv.lock bumps langchain to a profiles-capable version; test_model_fallback_resolution also conflicts. Port with/after #1778. | context-usage-ui | | `9bbd65d3` | #1781 | fix: derive model context windows from LangChain profiles (#1781) | Deferred | Pair with deferred #1778 on context-usage-ui: derives model context windows from LangChain profiles. options.py conflicts — re-key to the fork's Bedrock/Fireworks model map (upstream IDs differ; verify LangChain profiles even resolve for the fork's Bedrock-style IDs, else keep static values). uv.lock bumps langchain to a profiles-capable version; test_model_fallback_resolution also conflicts. Port with/after #1778. | context-usage-ui |
| `75fb8b48` | #1786 | Fix PR creation guard shell bypasses (#1786) | Deferred | SECURITY priority, clean merge: closes shell-bypass holes in PullRequestCreationGuardMiddleware (nested 'bash -c' expansion to depth 3, quoted-executable normalization) — a guard this fork actively wires. Port promptly; ALSO mirror the nested-shell expansion into the fork-only PullRequestVerdictGuardMiddleware (pr_verdict_guard.py), which shares the naive shlex approach and has the same bypass shape. | feature/upstream-clean-batch-jul21 |
| `32e81f29` | #1788 | Fix: Fix Insecure Direct Object Reference in slack_start_new_thread.py (#1788) | Deferred | SECURITY priority, clean merge: IDOR fix — slack_start_new_thread now enforces the deployment allowlist (_is_repo_allowed) and per-user repo access (require_repo_access_for_user keyed on the parent thread's github_login) before dispatching a run against a different repo. Both helpers exist in the fork at the same paths; merges clean. Port promptly. | feature/upstream-clean-batch-jul21 |
| `ab85b372` | #1764 | fix: add exc_info to swallowed exception in push re-review webhook (#1764) | Deferred | Trivial: adds exc_info=True to the swallowed exception log in process_github_push_event (webhooks/github.py). Merges clean. Batch with the next clean-port round. | feature/upstream-clean-batch-jul21 |
| `7312851d` | #1777 | feat: add explicit plan approval tool (#1777) | Deferred | Feature: explicit approve_plan tool + plan-mode exit. All deps exist in fork (plan_store PLAN_STATUS_APPROVED/SHARED, thread_api._user_owns_thread). Conflicts: prompt.py (fork prompt constants), server.py (tool/middleware wiring), check_message_queue.py (fork dashboard-handoff + mid-run injection drift), AGENTS.md. Hand re-key those four; keep the whole vertical (tool + plan_mode + thread_api + 4 test files) on one side. | | | `7312851d` | #1777 | feat: add explicit plan approval tool (#1777) | Deferred | Feature: explicit approve_plan tool + plan-mode exit. All deps exist in fork (plan_store PLAN_STATUS_APPROVED/SHARED, thread_api._user_owns_thread). Conflicts: prompt.py (fork prompt constants), server.py (tool/middleware wiring), check_message_queue.py (fork dashboard-handoff + mid-run injection drift), AGENTS.md. Hand re-key those four; keep the whole vertical (tool + plan_mode + thread_api + 4 test files) on one side. | |
| `e51abe14` | #1754 | fix: skip oversized images before model calls (#1754) | Deferred | Desirable: 10MB image size cap + 'image omitted' text block before model calls, prevents oversized-image model failures. Conflicts only because the fork hardened fetch_image_block (SSRF redirect guard, host-only logging) — hunks are compatible; re-key the size check around the fork's guarded fetch and port impl + test together. | | | `e51abe14` | #1754 | fix: skip oversized images before model calls (#1754) | Deferred | Desirable: 10MB image size cap + 'image omitted' text block before model calls, prevents oversized-image model failures. Conflicts only because the fork hardened fetch_image_block (SSRF redirect guard, host-only logging) — hunks are compatible; re-key the size check around the fork's guarded fetch and port impl + test together. | |
| `0ed560a5` | #1779 | fix: settle review checks after run failures (#1779) | Deferred | Desirable: run-failure completion webhook now settles reviewer check runs left open when the graph dies (_settle_failed_reviewer_check; uses settle_review_check_run + review_check_pending_result, both already in the fork's review/publish.py from the #214/#215 ports). Conflict is fork drift in completion.py (failure-reply customizations); re-key the new helper in and port with its 159-line test. | | | `0ed560a5` | #1779 | fix: settle review checks after run failures (#1779) | Deferred | Desirable: run-failure completion webhook now settles reviewer check runs left open when the graph dies (_settle_failed_reviewer_check; uses settle_review_check_run + review_check_pending_result, both already in the fork's review/publish.py from the #214/#215 ports). Conflict is fork drift in completion.py (failure-reply customizations); re-key the new helper in and port with its 159-line test. | |
| `bedc57b8` | #1796 | fix: cap execute output by making SandboxBackendProxy a BaseSandbox (#1796) | Deferred | Real fix (unbounded execute stdout pulled into the worker → OOM risk) but inapplicable at the fork's deepagents==0.6.12: no capture-offload API exists (ExecuteOffloadResult / execute_accepts_timeout absent; FilesystemMiddleware has no _resolve_capture BaseSandbox gate), so the bug it fixes cannot occur yet. Re-triage together with deferred #1745 (deepagents 0.6.12→0.7.x bump) — landing that bump makes this fix required. Fork's SandboxBackendProxy has diverged (sync-era, bound_repo repo-binding, no reconnect machinery): hand-apply the execute_with_offload/aexecute_with_offload delegation onto the fork proxy rather than cherry-pick. | | | `bedc57b8` | #1796 | fix: cap execute output by making SandboxBackendProxy a BaseSandbox (#1796) | Deferred | Real fix (unbounded execute stdout pulled into the worker → OOM risk) but inapplicable at the fork's deepagents==0.6.12: no capture-offload API exists (ExecuteOffloadResult / execute_accepts_timeout absent; FilesystemMiddleware has no _resolve_capture BaseSandbox gate), so the bug it fixes cannot occur yet. Re-triage together with deferred #1745 (deepagents 0.6.12→0.7.x bump) — landing that bump makes this fix required. Fork's SandboxBackendProxy has diverged (sync-era, bound_repo repo-binding, no reconnect machinery): hand-apply the execute_with_offload/aexecute_with_offload delegation onto the fork proxy rather than cherry-pick. | |
| `a77c4e47` | #1799 | fix: show current shared thread in sidebar (#1799) | Deferred | Desirable dashboard fix: sidebar now surfaces the currently-open shared thread (org-member 'Open in Web' links) when it falls outside the caller's own list — _sidebar_active_thread_summary + active_thread_id param; fork has the same include_all/org-member read path in thread_api.py. Cherry-picks CLEAN onto dev (merge-tree probe: zero conflicts); port with its thread-api tests + sandbox_id e2e spec + AgentsSidebar/queries UI as one vertical. | |
| `b67215ed` | #1780 | feat: link originating Slack threads (#1780) | Deferred | Desirable for the Slack-heavy fork: dispatch records the originating Slack permalink in thread metadata (webhooks/common.py), open_pull_request embeds an 'Originating Slack thread' link in the PR body (gated to private repos), and the sidebar menu links it. Impl files auto-merge (thread_api / open_pull_request / webhooks/common); conflicts are fork drift in prompt.py (customized prompt constants), tests/e2e/harness.py, tests/slack/test_slack_context.py — hand re-key those; keep backend + UI (AgentsSidebar, types) + e2e dashboard spec as one vertical. | | | `b67215ed` | #1780 | feat: link originating Slack threads (#1780) | Deferred | Desirable for the Slack-heavy fork: dispatch records the originating Slack permalink in thread metadata (webhooks/common.py), open_pull_request embeds an 'Originating Slack thread' link in the PR body (gated to private repos), and the sidebar menu links it. Impl files auto-merge (thread_api / open_pull_request / webhooks/common); conflicts are fork drift in prompt.py (customized prompt constants), tests/e2e/harness.py, tests/slack/test_slack_context.py — hand re-key those; keep backend + UI (AgentsSidebar, types) + e2e dashboard spec as one vertical. | |
| `7d79d300` | #1804 | chore(deps): bump deepagents to 0.7.0b1 pin (#1804) | Deferred | Rides the deferred #1745 deepagents bump: 0.7.0a7→0.7.0b1 pin + wcmatch>=11.0 uv override (langchain-e2b's e2b dep still caps wcmatch<11.0). Fork is pinned deepagents==0.6.12 — and langsmith==0.10.10 already satisfies the new >=0.10.9 floor — and never took the a7-era HARNESS_EXCLUDED_MIDDLEWARE code its prompt.py hunk deletes, so nothing applies standalone. Re-triage as one cluster: #1745 (bump, now targeting 0.7.0b1) + #1796 (offload delegation) + this pin/override. | | | `7d79d300` | #1804 | chore(deps): bump deepagents to 0.7.0b1 pin (#1804) | Deferred | Rides the deferred #1745 deepagents bump: 0.7.0a7→0.7.0b1 pin + wcmatch>=11.0 uv override (langchain-e2b's e2b dep still caps wcmatch<11.0). Fork is pinned deepagents==0.6.12 — and langsmith==0.10.10 already satisfies the new >=0.10.9 floor — and never took the a7-era HARNESS_EXCLUDED_MIDDLEWARE code its prompt.py hunk deletes, so nothing applies standalone. Re-triage as one cluster: #1745 (bump, now targeting 0.7.0b1) + #1796 (offload delegation) + this pin/override. | |
| `0e2eefbc` | #1801 | chore: bump default sandbox resources and delete-after-stop (#1801) | Deferred | Capacity/cost policy, not a correctness fix: defaults 2→4 vCPU, 7936MiB→16GiB mem, 32→128GiB FS, stopped-sandbox delete 24h→14d. Would materially raise LangSmith sandbox spend on sh-openswe unless prod env already overrides; all five values are env-overridable today. Decide prod sizing first (ops decision), then adopt or reject. Textual conflict is only fork drift in integrations/langsmith.py (sync-era SandboxClient, reconnect constants absent) — the constants re-key trivially; docs hunks auto-merge. | | | `0e2eefbc` | #1801 | chore: bump default sandbox resources and delete-after-stop (#1801) | Deferred | Capacity/cost policy, not a correctness fix: defaults 2→4 vCPU, 7936MiB→16GiB mem, 32→128GiB FS, stopped-sandbox delete 24h→14d. Would materially raise LangSmith sandbox spend on sh-openswe unless prod env already overrides; all five values are env-overridable today. Decide prod sizing first (ops decision), then adopt or reject. Textual conflict is only fork drift in integrations/langsmith.py (sync-era SandboxClient, reconnect constants absent) — the constants re-key trivially; docs hunks auto-merge. | |

View file

@ -214,6 +214,14 @@ def test_save_plan_exported_and_wired() -> None:
assert callable(save_plan) assert callable(save_plan)
def test_save_plan_description_warns_about_slack_images() -> None:
from agent.tools import save_plan
description = save_plan.__doc__ or ""
assert "persist Markdown text only" in description
assert "post them directly in Slack" in description
def test_plan_status_constants() -> None: def test_plan_status_constants() -> None:
from agent.dashboard import plan_store from agent.dashboard import plan_store

View file

@ -1360,6 +1360,161 @@ async def test_list_dashboard_threads_sidebar_fills_buckets_with_one_endpoint(mo
assert {call["offset"] for call in searches} == {0, page_size} assert {call["offset"] for call in searches} == {0, page_size}
async def test_list_dashboard_threads_sidebar_includes_readable_active_thread(
monkeypatch,
) -> None:
threads = _make_threads(1, resolved_before=0)
shared_thread = {
"thread_id": "shared-thread",
"metadata": {
"source": "slack",
"github_login": "teammate",
"title": "Teammate thread",
"updated_at_ms": 100,
"latest_run_status": "success",
"sandbox_id": "sandbox-123",
},
}
class FakeThreads:
async def search(self, *, metadata, limit, offset, sort_by, sort_order, select):
assert select == thread_api._THREAD_LIST_SELECT
return threads[offset : offset + limit]
async def get(self, thread_id):
assert thread_id == "shared-thread"
return shared_thread
async def update(self, *, thread_id, metadata):
return None
class FakeRuns:
async def list(self, thread_id, limit=1):
return []
class FakeClient:
threads = FakeThreads()
runs = FakeRuns()
monkeypatch.setattr(thread_api, "langgraph_client", lambda: FakeClient())
result = await thread_api.list_dashboard_threads_sidebar(
"octocat",
email=None,
active_limit=5,
resolved_limit=5,
active_thread_id="shared-thread",
)
assert [item["id"] for item in result["active"]["items"]] == ["shared-thread", "t0"]
shared = result["active"]["items"][0]
assert shared["isOwner"] is False
assert shared["sandboxId"] == "sandbox-123"
async def test_list_dashboard_threads_sidebar_keeps_resolved_active_thread_resolved(
monkeypatch,
) -> None:
threads = _make_threads(1, resolved_before=0)
shared_thread = {
"thread_id": "shared-resolved-thread",
"metadata": {
"source": "slack",
"github_login": "teammate",
"title": "Resolved teammate thread",
"updated_at_ms": 100,
"latest_run_status": "success",
"resolved": True,
"sandbox_id": "sandbox-456",
},
}
class FakeThreads:
async def search(self, *, metadata, limit, offset, sort_by, sort_order, select):
assert select == thread_api._THREAD_LIST_SELECT
return threads[offset : offset + limit]
async def get(self, thread_id):
assert thread_id == "shared-resolved-thread"
return shared_thread
async def update(self, *, thread_id, metadata):
return None
class FakeRuns:
async def list(self, thread_id, limit=1):
return []
class FakeClient:
threads = FakeThreads()
runs = FakeRuns()
monkeypatch.setattr(thread_api, "langgraph_client", lambda: FakeClient())
result = await thread_api.list_dashboard_threads_sidebar(
"octocat",
email=None,
active_limit=5,
resolved_limit=5,
active_thread_id="shared-resolved-thread",
)
assert [item["id"] for item in result["active"]["items"]] == ["t0"]
assert [item["id"] for item in result["resolved"]["items"]] == ["shared-resolved-thread"]
shared = result["resolved"]["items"][0]
assert shared["isOwner"] is False
assert shared["resolved"] is True
assert shared["sandboxId"] == "sandbox-456"
async def test_list_dashboard_threads_sidebar_ignores_unreadable_active_thread(
monkeypatch,
) -> None:
threads = _make_threads(1, resolved_before=0)
private_thread = {
"thread_id": "private-thread",
"metadata": {
"source": "internal",
"github_login": "teammate",
"title": "Private thread",
"updated_at_ms": 100,
"latest_run_status": "success",
},
}
class FakeThreads:
async def search(self, *, metadata, limit, offset, sort_by, sort_order, select):
assert select == thread_api._THREAD_LIST_SELECT
return threads[offset : offset + limit]
async def get(self, thread_id):
assert thread_id == "private-thread"
return private_thread
async def update(self, *, thread_id, metadata):
return None
class FakeRuns:
async def list(self, thread_id, limit=1):
return []
class FakeClient:
threads = FakeThreads()
runs = FakeRuns()
monkeypatch.setattr(thread_api, "langgraph_client", lambda: FakeClient())
result = await thread_api.list_dashboard_threads_sidebar(
"octocat",
email=None,
active_limit=5,
resolved_limit=5,
active_thread_id="private-thread",
)
assert [item["id"] for item in result["active"]["items"]] == ["t0"]
async def test_list_dashboard_threads_page_refreshes_only_unsettled_threads(monkeypatch) -> None: async def test_list_dashboard_threads_page_refreshes_only_unsettled_threads(monkeypatch) -> None:
threads = _make_threads(3, resolved_before=0) threads = _make_threads(3, resolved_before=0)
threads[0]["metadata"]["latest_run_status"] = "success" threads[0]["metadata"]["latest_run_status"] = "success"

View file

@ -5,6 +5,7 @@ import { test, expect, type Locator, type Page } from "@playwright/test";
// creates a sandbox and stamps its id into the thread metadata, and the UI // creates a sandbox and stamps its id into the thread metadata, and the UI
// copies that same id. Only the LLM/GitHub/Slack boundaries are faked. // copies that same id. Only the LLM/GitHub/Slack boundaries are faked.
const SAME_USER = { login: "alice", email: "alice@example.com" }; const SAME_USER = { login: "alice", email: "alice@example.com" };
const OTHER_USER = { login: "bob", email: "bob@example.com" };
async function loginAs(page: Page, user: { login: string; email: string }) { async function loginAs(page: Page, user: { login: string; email: string }) {
const res = await page.request.post("/control/login", { data: user }); const res = await page.request.post("/control/login", { data: user });
@ -112,4 +113,39 @@ test.describe("thread sandbox id (real dashboard UI)", () => {
await context.close(); await context.close();
}); });
test("shared active thread appears in the sidebar with sandbox action", async ({
page,
browser,
baseURL,
}, testInfo) => {
await loginAs(page, SAME_USER);
const threadId = await createThreadWithSandbox(page);
const bobContext = await browser.newContext({ baseURL });
await bobContext.grantPermissions(["clipboard-read", "clipboard-write"], {
origin: baseURL,
});
const bobPage = await bobContext.newPage();
await loginAs(bobPage, OTHER_USER);
await bobPage.goto(`/agents/${threadId}`);
await expect(bobPage).toHaveURL(new RegExp(`/agents/${threadId}$`));
const row = bobPage.locator(`a[href$="/agents/${threadId}"]`).first();
await expect(row).toBeVisible();
await row.hover();
await kebabFor(row).click();
await expect(copyItem(bobPage)).toBeEnabled();
const screenshotPath = testInfo.outputPath(
"shared-thread-sidebar-sandbox-menu.png",
);
await bobPage.screenshot({ path: screenshotPath, fullPage: true });
await testInfo.attach("shared-thread-sidebar-sandbox-menu", {
path: screenshotPath,
contentType: "image/png",
});
await bobContext.close();
});
}); });

View file

@ -38,6 +38,23 @@ def test_detects_pr_creation_fallback_commands() -> None:
assert is_pr_creation_fallback_command( assert is_pr_creation_fallback_command(
"curl -X POST https://api.github.com/repos/langchain-ai/open-swe/pulls -d '{}'" "curl -X POST https://api.github.com/repos/langchain-ai/open-swe/pulls -d '{}'"
) )
assert is_pr_creation_fallback_command("/usr/bin/gh pr create --draft")
assert is_pr_creation_fallback_command(
"/usr/bin/curl -X POST https://api.github.com/repos/langchain-ai/open-swe/pulls -d '{}'"
)
assert is_pr_creation_fallback_command("bash -c 'gh pr create --draft'")
assert is_pr_creation_fallback_command("bash -c'gh pr create --draft'")
assert is_pr_creation_fallback_command("zsh -lc'gh pr create --draft'")
assert is_pr_creation_fallback_command("GH_TOKEN=dummy sh -c 'gh pr create --draft'")
assert is_pr_creation_fallback_command(
"zsh -lc 'gh api repos/langchain-ai/open-swe/pulls -X POST -f title=x'"
)
assert is_pr_creation_fallback_command(
"bash -c \"curl -X POST https://api.github.com/repos/langchain-ai/open-swe/pulls -d '{}'\""
)
assert is_pr_creation_fallback_command(
'bash -c \'sh -c "dash -c \\"zsh -c \\\\\\"gh pr create --draft\\\\\\"\\""\''
)
def test_allows_safe_pr_commands() -> None: def test_allows_safe_pr_commands() -> None:
@ -45,6 +62,8 @@ def test_allows_safe_pr_commands() -> None:
assert not is_pr_creation_fallback_command("gh pr list --head open-swe/foo") assert not is_pr_creation_fallback_command("gh pr list --head open-swe/foo")
assert not is_pr_creation_fallback_command("gh pr edit 1 --add-label ready") assert not is_pr_creation_fallback_command("gh pr edit 1 --add-label ready")
assert not is_pr_creation_fallback_command("gh pr comment 1 --body done") assert not is_pr_creation_fallback_command("gh pr comment 1 --body done")
assert not is_pr_creation_fallback_command("bash -c 'gh pr view 1 --json url'")
assert not is_pr_creation_fallback_command("/usr/bin/gh pr view 1 --json url")
async def test_middleware_blocks_execute_pr_creation_fallbacks() -> None: async def test_middleware_blocks_execute_pr_creation_fallbacks() -> None:

View file

@ -60,6 +60,28 @@ def test_detects_curl_review_verdicts() -> None:
) )
def test_detects_nested_and_normalized_verdict_commands() -> None:
assert is_pr_verdict_fallback_command("/usr/bin/gh pr review 12 --approve")
assert is_pr_verdict_fallback_command(
"/usr/bin/curl -X POST https://api.github.com/repos/o/r/pulls/5/reviews "
'-d \'{"event": "APPROVE"}\''
)
assert is_pr_verdict_fallback_command("bash -c 'gh pr review 12 --approve'")
assert is_pr_verdict_fallback_command("bash -c'gh pr review 12 --approve'")
assert is_pr_verdict_fallback_command("zsh -lc'gh pr review 12 --approve'")
assert is_pr_verdict_fallback_command("GH_TOKEN=dummy sh -c 'gh pr review -a'")
assert is_pr_verdict_fallback_command(
"zsh -lc 'gh api repos/o/r/pulls/5/reviews -X POST -f event=REQUEST_CHANGES'"
)
assert is_pr_verdict_fallback_command(
'bash -c "curl -X POST https://api.github.com/repos/o/r/pulls/5/reviews '
'-d \'{\\"event\\": \\"APPROVE\\"}\'"'
)
assert is_pr_verdict_fallback_command(
'bash -c \'sh -c "dash -c \\"zsh -c \\\\\\"gh pr review 12 --approve\\\\\\"\\""\''
)
def test_allows_safe_review_commands() -> None: def test_allows_safe_review_commands() -> None:
assert not is_pr_verdict_fallback_command("gh pr review 12 --comment -b 'looks good'") assert not is_pr_verdict_fallback_command("gh pr review 12 --comment -b 'looks good'")
assert not is_pr_verdict_fallback_command("gh pr review -c") assert not is_pr_verdict_fallback_command("gh pr review -c")
@ -69,6 +91,8 @@ def test_allows_safe_review_commands() -> None:
assert not is_pr_verdict_fallback_command("gh api repos/o/r/pulls/5/reviews") assert not is_pr_verdict_fallback_command("gh api repos/o/r/pulls/5/reviews")
assert not is_pr_verdict_fallback_command("gh pr create --draft") assert not is_pr_verdict_fallback_command("gh pr create --draft")
assert not is_pr_verdict_fallback_command("curl https://api.github.com/repos/o/r/pulls/5") assert not is_pr_verdict_fallback_command("curl https://api.github.com/repos/o/r/pulls/5")
assert not is_pr_verdict_fallback_command("bash -c 'gh pr review 12 --comment -b ok'")
assert not is_pr_verdict_fallback_command("/usr/bin/gh pr view 12 --json reviews")
async def test_middleware_blocks_execute_verdict_fallbacks() -> None: async def test_middleware_blocks_execute_verdict_fallbacks() -> None:

View file

@ -111,7 +111,7 @@ export function AgentsSidebar({
}: AgentsSidebarProps) { }: AgentsSidebarProps) {
const { prefs, setGroup, setCompact, setFilters, resetFilters } = const { prefs, setGroup, setCompact, setFilters, resetFilters } =
useSidebarPrefs() useSidebarPrefs()
const sidebar = useSidebarThreads(RESOLVED_SIDEBAR_LIMIT) const sidebar = useSidebarThreads(RESOLVED_SIDEBAR_LIMIT, activeThreadId)
const activeThreads = sidebar.data?.active.items ?? [] const activeThreads = sidebar.data?.active.items ?? []
const resolvedThreads = sidebar.data?.resolved.items ?? [] const resolvedThreads = sidebar.data?.resolved.items ?? []
const resolvedHasMore = sidebar.data?.resolved.hasMore ?? false const resolvedHasMore = sidebar.data?.resolved.hasMore ?? false

View file

@ -225,6 +225,7 @@ function buildThreadsPageQuery(params: ThreadsPageParams): string {
function buildSidebarThreadsQuery(params: { function buildSidebarThreadsQuery(params: {
activeLimit?: number activeLimit?: number
resolvedLimit?: number resolvedLimit?: number
activeThreadId?: string
}): string { }): string {
const search = new URLSearchParams() const search = new URLSearchParams()
if (params.activeLimit != null) { if (params.activeLimit != null) {
@ -233,6 +234,9 @@ function buildSidebarThreadsQuery(params: {
if (params.resolvedLimit != null) { if (params.resolvedLimit != null) {
search.set("resolved_limit", String(params.resolvedLimit)) search.set("resolved_limit", String(params.resolvedLimit))
} }
if (params.activeThreadId) {
search.set("active_thread_id", params.activeThreadId)
}
const query = search.toString() const query = search.toString()
return query ? `?${query}` : "" return query ? `?${query}` : ""
} }
@ -242,6 +246,7 @@ export const agentsApi = {
listSidebarThreads: (params: { listSidebarThreads: (params: {
activeLimit?: number activeLimit?: number
resolvedLimit?: number resolvedLimit?: number
activeThreadId?: string
}) => }) =>
agentsRequest<SidebarThreads>( agentsRequest<SidebarThreads>(
`/threads/sidebar${buildSidebarThreadsQuery(params)}` `/threads/sidebar${buildSidebarThreadsQuery(params)}`

View file

@ -13,8 +13,11 @@ import type { AgentThread, Chunk, ImageChunk, Message } from "./types"
export const agentThreadKeys = { export const agentThreadKeys = {
lists: ["agent-threads", "lists"] as const, lists: ["agent-threads", "lists"] as const,
sidebar: (params: { activeLimit: number; resolvedLimit: number }) => sidebar: (params: {
["agent-threads", "lists", "sidebar", params] as const, activeLimit: number
resolvedLimit: number
activeThreadId?: string
}) => ["agent-threads", "lists", "sidebar", params] as const,
detail: (threadId: string) => ["agent-threads", threadId] as const, detail: (threadId: string) => ["agent-threads", threadId] as const,
prDiff: (threadId: string) => ["agent-threads", threadId, "pr-diff"] as const, prDiff: (threadId: string) => ["agent-threads", threadId, "pr-diff"] as const,
workflowApprovals: (threadId: string) => workflowApprovals: (threadId: string) =>
@ -93,10 +96,14 @@ function sidebarRefetchInterval(query: { state: { data?: SidebarThreads } }) {
: false : false
} }
export function useSidebarThreads(resolvedLimit: number) { export function useSidebarThreads(
resolvedLimit: number,
activeThreadId?: string
) {
const params = { const params = {
activeLimit: SIDEBAR_ACTIVE_LIMIT, activeLimit: SIDEBAR_ACTIVE_LIMIT,
resolvedLimit, resolvedLimit,
activeThreadId,
} }
return useQuery({ return useQuery({
queryKey: agentThreadKeys.sidebar(params), queryKey: agentThreadKeys.sidebar(params),

View file

@ -1262,7 +1262,12 @@ function ReviewBodyInner({
deletions: detail.pr.deletions, deletions: detail.pr.deletions,
}} }}
/> />
<div className="mt-4 rounded-lg border border-border bg-card p-4"> <div
className={cn(
"mt-4 rounded-lg border border-border p-4",
embedded ? "bg-[var(--ui-surface)]" : "bg-card"
)}
>
{detail.pr.body ? ( {detail.pr.body ? (
<Markdown <Markdown
content={detail.pr.body} content={detail.pr.body}