diff --git a/agent/dashboard/routes.py b/agent/dashboard/routes.py index 92d34594..448ef0ff 100644 --- a/agent/dashboard/routes.py +++ b/agent/dashboard/routes.py @@ -1553,6 +1553,7 @@ async def api_list_threads( async def api_list_threads_sidebar( active_limit: int = 50, resolved_limit: int = 20, + active_thread_id: str | None = None, all: bool = False, session: dict[str, Any] = _SESSION_DEP, ) -> dict[str, Any]: @@ -1563,6 +1564,7 @@ async def api_list_threads_sidebar( email=session.get("email"), active_limit=active_limit, resolved_limit=resolved_limit, + active_thread_id=active_thread_id, include_all=all, ) diff --git a/agent/dashboard/thread_api.py b/agent/dashboard/thread_api.py index cc3f3015..aae4bf2b 100644 --- a/agent/dashboard/thread_api.py +++ b/agent/dashboard/thread_api.py @@ -8,7 +8,7 @@ import binascii import json import logging import os -from collections.abc import AsyncIterator +from collections.abc import AsyncIterator, Mapping from datetime import UTC, datetime from typing import Any @@ -722,12 +722,58 @@ async def list_dashboard_threads( return page["items"] +async def _sidebar_active_thread_summary( + client: Any, + active_thread_id: str | None, + *, + fallback_threads: Mapping[str, dict[str, Any]], + visible_thread_ids: set[str], + login: str, + email: str | None, + include_all: bool, +) -> tuple[dict[str, Any], bool] | None: + if not active_thread_id or active_thread_id in visible_thread_ids: + return None + thread = fallback_threads.get(active_thread_id) + if thread is None: + try: + fetched = await client.threads.get(active_thread_id) + except Exception: # noqa: BLE001 + logger.debug( + "Could not fetch active sidebar thread %s", active_thread_id, exc_info=True + ) + return None + if not isinstance(fetched, Mapping): + return None + thread = fetched + metadata = _thread_metadata(thread) + if not include_all: + try: + _assert_thread_readable(metadata) + except HTTPException: + return None + # Refreshing persists latest-run metadata back to the thread; only do that + # when the caller owns it, mirroring the is_owner gate on the single-thread + # read path. A non-owner viewing a shared thread reads its last-known state + # without mutating it. + owns_thread = _user_owns_thread(metadata, login, email) + summary = await _summarize_thread( + client, + thread, + owner_login=None if include_all else login, + owner_email=None if include_all else email, + refresh_active_run=owns_thread, + ) + return summary, _is_thread_resolved(metadata) + + async def list_dashboard_threads_sidebar( login: str, *, email: str | None = None, active_limit: int = 50, resolved_limit: int = 20, + active_thread_id: str | None = None, include_all: bool = False, ) -> dict[str, Any]: client = langgraph_client() @@ -775,7 +821,8 @@ async def list_dashboard_threads_sidebar( resolved_candidates = sorted(resolved_threads.values(), key=_thread_updated_ms, reverse=True) active_window = active_candidates[:safe_active_limit] resolved_window = resolved_candidates[:safe_resolved_limit] - active_items, resolved_items = await asyncio.gather( + active_ids = {thread_id for thread in active_window if (thread_id := _thread_id(thread))} + active_items, resolved_items, active_thread = await asyncio.gather( _summarize_threads( client, active_window, @@ -788,17 +835,52 @@ async def list_dashboard_threads_sidebar( owner_login=None if include_all else login, owner_email=None if include_all else email, ), + _sidebar_active_thread_summary( + client, + active_thread_id, + fallback_threads={**active, **resolved_threads}, + visible_thread_ids=active_ids, + login=login, + email=email, + include_all=include_all, + ), ) + active_has_more = len(active_candidates) > safe_active_limit + resolved_has_more = len(resolved_candidates) > safe_resolved_limit + if active_thread: + active_thread_summary, is_resolved_active_thread = active_thread + if is_resolved_active_thread: + resolved_items = [ + active_thread_summary, + *[item for item in resolved_items if item["id"] != active_thread_summary["id"]], + ] + active_items = [ + item for item in active_items if item["id"] != active_thread_summary["id"] + ] + if len(resolved_items) > safe_resolved_limit: + resolved_items = resolved_items[:safe_resolved_limit] + resolved_has_more = True + else: + active_items = [ + active_thread_summary, + *[item for item in active_items if item["id"] != active_thread_summary["id"]], + ] + resolved_items = [ + item for item in resolved_items if item["id"] != active_thread_summary["id"] + ] + if len(active_items) > safe_active_limit: + active_items = active_items[:safe_active_limit] + active_has_more = True return { "active": { "items": active_items, "limit": safe_active_limit, - "hasMore": len(active_candidates) > safe_active_limit, + "hasMore": active_has_more, }, "resolved": { "items": resolved_items, "limit": safe_resolved_limit, - "hasMore": len(resolved_candidates) > safe_resolved_limit, + "hasMore": resolved_has_more, }, } diff --git a/agent/middleware/pr_creation_guard.py b/agent/middleware/pr_creation_guard.py index 3420cfae..2541356a 100644 --- a/agent/middleware/pr_creation_guard.py +++ b/agent/middleware/pr_creation_guard.py @@ -3,6 +3,7 @@ from __future__ import annotations import json +import os import re import shlex from collections.abc import Awaitable, Callable, Mapping @@ -14,6 +15,9 @@ from langgraph.prebuilt.tool_node import ToolCallRequest from langgraph.types import Command _SHELL_SEPARATORS = {";", "&&", "||", "|", "&"} +_SHELL_EXECUTABLES = {"bash", "dash", "sh", "zsh"} +_MAX_SHELL_EXPANSION_DEPTH = 3 +_SHELL_EXPANSION_DEPTH_LIMIT_TOKEN = "__pr_creation_guard_shell_expansion_depth_limit__" _GITHUB_PULLS_ENDPOINT = re.compile(r"(?:^|/)repos/[^/\s]+/[^/\s]+/pulls/?$") _GITHUB_PULLS_URL = re.compile(r"https://api\.github\.com/repos/[^/\s]+/[^/\s]+/pulls/?") _BLOCK_ERROR = ( @@ -46,13 +50,62 @@ def _tool_call_id(request: ToolCallRequest) -> str | None: return None -def _shell_tokens(command: str) -> list[str]: +def _split_shell_tokens(command: str) -> list[str]: try: return shlex.split(command, posix=True) except ValueError: return command.split() +def _executable_name(token: str) -> str: + return os.path.basename(token.strip("'\"")) + + +def _shell_command_argument(tokens: list[str], shell_index: int) -> str | None: + for index, token in enumerate(tokens[shell_index + 1 :], start=shell_index + 1): + if token in _SHELL_SEPARATORS: + return None + if token == "-c" or ( + token.startswith("-") and not token.startswith("--") and "c" in token[1:] + ): + _, _, glued = token[1:].partition("c") + if glued: + return glued + if index + 1 < len(tokens) and tokens[index + 1] not in _SHELL_SEPARATORS: + return tokens[index + 1] + return None + return None + + +def _has_nested_shell_command(tokens: list[str]) -> bool: + return any( + _executable_name(token) in _SHELL_EXECUTABLES + and _shell_command_argument(tokens, index) is not None + for index, token in enumerate(tokens) + ) + + +def _expand_nested_shell_tokens(tokens: list[str], depth: int = 0) -> list[str]: + if depth >= _MAX_SHELL_EXPANSION_DEPTH: + if _has_nested_shell_command(tokens): + return [*tokens, _SHELL_EXPANSION_DEPTH_LIMIT_TOKEN] + return tokens + + expanded = list(tokens) + for index, token in enumerate(tokens): + if _executable_name(token) not in _SHELL_EXECUTABLES: + continue + inner_command = _shell_command_argument(tokens, index) + if inner_command is None: + continue + expanded.extend(_expand_nested_shell_tokens(_split_shell_tokens(inner_command), depth + 1)) + return expanded + + +def _shell_tokens(command: str) -> list[str]: + return _expand_nested_shell_tokens(_split_shell_tokens(command)) + + def _is_assignment(token: str) -> bool: name, sep, _value = token.partition("=") return bool(sep and name and re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", name)) @@ -69,7 +122,7 @@ def _gh_subtokens(tokens: list[str], index: int) -> list[str]: def _contains_gh_pr_create(tokens: list[str]) -> bool: for index, token in enumerate(tokens): - if token != "gh": + if _executable_name(token) != "gh": continue subtokens = _gh_subtokens(tokens, index) for offset, subtoken in enumerate(subtokens[:-1]): @@ -136,7 +189,7 @@ def _gh_api_uses_post_or_body(subtokens: list[str]) -> bool: def _contains_gh_api_pull_create(tokens: list[str]) -> bool: for index, token in enumerate(tokens): - if token != "gh": + if _executable_name(token) != "gh": continue subtokens = _gh_subtokens(tokens, index) if "api" not in subtokens: @@ -153,7 +206,7 @@ def _contains_gh_api_pull_create(tokens: list[str]) -> bool: def _contains_direct_pull_create(tokens: list[str]) -> bool: for index, token in enumerate(tokens): - if token != "curl": + if _executable_name(token) != "curl": continue subtokens: list[str] = [] for candidate in tokens[index + 1 :]: @@ -193,7 +246,8 @@ def is_pr_creation_fallback_command(command: str) -> bool: """ tokens = _shell_tokens(command) return ( - _contains_gh_pr_create(tokens) + _SHELL_EXPANSION_DEPTH_LIMIT_TOKEN in tokens + or _contains_gh_pr_create(tokens) or _contains_gh_api_pull_create(tokens) or _contains_direct_pull_create(tokens) ) diff --git a/agent/middleware/pr_verdict_guard.py b/agent/middleware/pr_verdict_guard.py index 78ebbd90..ad95883c 100644 --- a/agent/middleware/pr_verdict_guard.py +++ b/agent/middleware/pr_verdict_guard.py @@ -3,6 +3,7 @@ from __future__ import annotations import json +import os import re import shlex from collections.abc import Awaitable, Callable, Mapping @@ -14,6 +15,9 @@ from langgraph.prebuilt.tool_node import ToolCallRequest from langgraph.types import Command _SHELL_SEPARATORS = {";", "&&", "||", "|", "&"} +_SHELL_EXECUTABLES = {"bash", "dash", "sh", "zsh"} +_MAX_SHELL_EXPANSION_DEPTH = 3 +_SHELL_EXPANSION_DEPTH_LIMIT_TOKEN = "__pr_verdict_guard_shell_expansion_depth_limit__" _GITHUB_REVIEWS_ENDPOINT = re.compile( r"(?:^|/)repos/[^/\s]+/[^/\s]+/pulls/\d+/reviews(?:/\d+/events)?/?$" ) @@ -55,13 +59,62 @@ def _tool_call_id(request: ToolCallRequest) -> str | None: return None -def _shell_tokens(command: str) -> list[str]: +def _split_shell_tokens(command: str) -> list[str]: try: return shlex.split(command, posix=True) except ValueError: return command.split() +def _executable_name(token: str) -> str: + return os.path.basename(token.strip("'\"")) + + +def _shell_command_argument(tokens: list[str], shell_index: int) -> str | None: + for index, token in enumerate(tokens[shell_index + 1 :], start=shell_index + 1): + if token in _SHELL_SEPARATORS: + return None + if token == "-c" or ( + token.startswith("-") and not token.startswith("--") and "c" in token[1:] + ): + _, _, glued = token[1:].partition("c") + if glued: + return glued + if index + 1 < len(tokens) and tokens[index + 1] not in _SHELL_SEPARATORS: + return tokens[index + 1] + return None + return None + + +def _has_nested_shell_command(tokens: list[str]) -> bool: + return any( + _executable_name(token) in _SHELL_EXECUTABLES + and _shell_command_argument(tokens, index) is not None + for index, token in enumerate(tokens) + ) + + +def _expand_nested_shell_tokens(tokens: list[str], depth: int = 0) -> list[str]: + if depth >= _MAX_SHELL_EXPANSION_DEPTH: + if _has_nested_shell_command(tokens): + return [*tokens, _SHELL_EXPANSION_DEPTH_LIMIT_TOKEN] + return tokens + + expanded = list(tokens) + for index, token in enumerate(tokens): + if _executable_name(token) not in _SHELL_EXECUTABLES: + continue + inner_command = _shell_command_argument(tokens, index) + if inner_command is None: + continue + expanded.extend(_expand_nested_shell_tokens(_split_shell_tokens(inner_command), depth + 1)) + return expanded + + +def _shell_tokens(command: str) -> list[str]: + return _expand_nested_shell_tokens(_split_shell_tokens(command)) + + def _subtokens_after(tokens: list[str], index: int) -> list[str]: subtokens: list[str] = [] for token in tokens[index + 1 :]: @@ -73,7 +126,7 @@ def _subtokens_after(tokens: list[str], index: int) -> list[str]: def _contains_gh_pr_review_verdict(tokens: list[str]) -> bool: for index, token in enumerate(tokens): - if token != "gh": + if _executable_name(token) != "gh": continue subtokens = _subtokens_after(tokens, index) is_pr_review = any( @@ -92,7 +145,7 @@ def _contains_gh_pr_review_verdict(tokens: list[str]) -> bool: def _contains_gh_api_review_verdict(tokens: list[str]) -> bool: for index, token in enumerate(tokens): - if token != "gh": + if _executable_name(token) != "gh": continue subtokens = _subtokens_after(tokens, index) if "api" not in subtokens: @@ -111,7 +164,7 @@ def _contains_gh_api_review_verdict(tokens: list[str]) -> bool: def _contains_direct_review_verdict(tokens: list[str]) -> bool: for index, token in enumerate(tokens): - if token != "curl": + if _executable_name(token) != "curl": continue subtokens = _subtokens_after(tokens, index) if not any(_GITHUB_REVIEWS_URL.search(subtoken) for subtoken in subtokens): @@ -129,7 +182,13 @@ def is_pr_verdict_fallback_command(command: str) -> bool: ``event=APPROVE|REQUEST_CHANGES`` to a ``/pulls/N/reviews`` endpoint, and ``curl`` to the reviews URL with a verdict event in the body) but is intentionally fail-open: shell aliases, ``gh`` aliases, ``--input`` - JSON-file bodies, and non-curl HTTP clients will not be blocked. This is + JSON-file bodies, and non-curl HTTP clients will not be blocked. The + ``-c`` form of a known shell (``bash``/``dash``/``sh``/``zsh``, whether + space-separated or glued as ``-c'...'``) is expanded to a fixed depth and + quoted/path-prefixed executables are normalized, so those do not slip + through; other shells (``ash``/``busybox``/``fish``) and stdin-fed + programs (``... | bash``, here-strings, ``bash -s``) remain fail-open. + This is acceptable because the threat model is an honest agent papering over a ``publish_review`` limitation or refusal, not an adversary trying to bypass the guardrail. Plain ``gh pr review`` (no verdict flag) and @@ -138,7 +197,8 @@ def is_pr_verdict_fallback_command(command: str) -> bool: """ tokens = _shell_tokens(command) return ( - _contains_gh_pr_review_verdict(tokens) + _SHELL_EXPANSION_DEPTH_LIMIT_TOKEN in tokens + or _contains_gh_pr_review_verdict(tokens) or _contains_gh_api_review_verdict(tokens) or _contains_direct_review_verdict(tokens) ) diff --git a/agent/tools/save_plan.py b/agent/tools/save_plan.py index ac7f4cbb..3f97061f 100644 --- a/agent/tools/save_plan.py +++ b/agent/tools/save_plan.py @@ -43,7 +43,10 @@ async def save_plan( shared content. Write the content in standard Markdown — headings, bullet/numbered lists, and - fenced code blocks all render. + fenced code blocks all render. Shared responses persist Markdown text only; + they do not upload or serve sandbox-local or relative image files. If images + or screenshots are needed for a Slack response, post them directly in Slack + instead of relying on the shared response page. Args: plan_file_path: Path to the Markdown plan file in the sandbox. diff --git a/agent/tools/slack_start_new_thread.py b/agent/tools/slack_start_new_thread.py index cf9de0ce..8dd2044d 100644 --- a/agent/tools/slack_start_new_thread.py +++ b/agent/tools/slack_start_new_thread.py @@ -2,9 +2,11 @@ import os import re from typing import Any +from fastapi import HTTPException from langgraph.config import get_config from langgraph_sdk import get_client +from ..dashboard.repo_access import require_repo_access_for_user from ..dispatch import dispatch_agent_run from ..utils.dashboard_links import dashboard_thread_url from ..utils.slack import ( @@ -13,6 +15,7 @@ from ..utils.slack import ( store_slack_run_mapping, ) from ..utils.thread_ids import generate_thread_id_from_slack_thread +from ..webhooks.common import _is_repo_allowed LANGGRAPH_URL = os.environ.get("LANGGRAPH_URL") or os.environ.get( "LANGGRAPH_URL_PROD", "http://localhost:2024" @@ -160,6 +163,42 @@ async def slack_start_new_thread( "error": "default_repo must be a simple owner/name repository string", } + if default_repo and default_repo.strip() and repo is not None: + if not _is_repo_allowed(repo): + return { + "success": False, + "error": ( + f"Repository {repo['owner']}/{repo['name']} is not on the deployment allowlist" + ), + } + github_login = configurable.get("github_login") + if not isinstance(github_login, str) or not github_login.strip(): + return { + "success": False, + "error": ( + "Cannot verify access to the requested repository: no github_login on the " + "parent thread" + ), + } + try: + await require_repo_access_for_user( + github_login.strip(), f"{repo['owner']}/{repo['name']}" + ) + except HTTPException as exc: + return { + "success": False, + "error": ( + f"Access to repository {repo['owner']}/{repo['name']} denied: {exc.detail}" + ), + } + except Exception as exc: # noqa: BLE001 + return { + "success": False, + "error": ( + f"Failed to verify access to repository {repo['owner']}/{repo['name']}: {exc}" + ), + } + message_ts, slack_error = await post_slack_top_level_message_with_ts( channel_id.strip(), _visible_message(clean_title, clean_instructions, repo), diff --git a/agent/webhooks/github.py b/agent/webhooks/github.py index 45f0d87f..36074959 100644 --- a/agent/webhooks/github.py +++ b/agent/webhooks/github.py @@ -589,7 +589,9 @@ async def process_github_push_event(payload: dict[str, Any]) -> None: await common.reconcile_findings_with_review_threads(thread_id, threads) except Exception: common.logger.warning( - "Could not sync review threads before push re-review for %s", thread_id + "Could not sync review threads before push re-review for %s", + thread_id, + exc_info=True, ) pr_meta: ReviewerPRMeta = { diff --git a/docs/upstream-sync/triage.jsonl b/docs/upstream-sync/triage.jsonl index 41ed31d8..a7c1aeae 100644 --- a/docs/upstream-sync/triage.jsonl +++ b/docs/upstream-sync/triage.jsonl @@ -130,14 +130,14 @@ {"sha": "f0897479", "pr": 1778, "subject": "feat: surface context window usage in agents UI (#1778)", "disposition": "deferred", "reason": "Near-clean: context-window indicator UI is all new files (zero drift); options.py hunk must be re-keyed to the fork's Bedrock/Fireworks model map (fork model IDs differ from upstream's) — add context_window per fork entry rather than taking upstream values.", "branch": "context-usage-ui", "local_sha": null, "updated": "2026-07-17T22:33:46Z"} {"sha": "31263f83", "pr": 1785, "subject": "Fix: Fix Stored XSS in ReplyCard.tsx (#1785)", "disposition": "landed", "reason": "SECURITY priority, near-clean: diff is urlencode() hardening of the GitHub OAuth authorize URL in dashboard routes.py auth_login (upstream bot title says ReplyCard.tsx — mismatch, trust the diff). Fork auth_login has the identical f-string URL; hunk applies clean. Port promptly.", "branch": "feature/upstream-clean-batch-security-linear", "local_sha": null, "updated": "2026-07-20T19:46:00Z"} {"sha": "3ea29d3f", "pr": 1789, "subject": "Fix: Fix Improper privilege management in server.py (#1789)", "disposition": "landed", "reason": "SECURITY priority, near-clean: adds empty-signature reject to verify_github_signature (utils/github_comments.py) + verify_linear_signature (webhooks/common.py) (title says server.py — mismatch, trust the diff). Both fork functions match at the hunk sites; fork-only verify_jira_secret already guards empty token. Real value: None signature currently raises TypeError in compare_digest. Port BOTH hunks together.", "branch": "feature/upstream-clean-batch-security-linear", "local_sha": null, "updated": "2026-07-20T19:46:00Z"} -{"sha": "2e8ff4b7", "pr": 1782, "subject": "chore: clarify shared response image guidance (#1782)", "disposition": "deferred", "reason": "Near-clean: docstring-only guidance in save_plan (shared responses persist Markdown only, never sandbox-local images — post screenshots directly to Slack) + 2 assertions in test_plan_review. Merges clean against fork. Batch with the next clean-port round.", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-21T18:00:13Z"} -{"sha": "4ea2441a", "pr": 1791, "subject": "fix: match embedded review description background (#1791)", "disposition": "deferred", "reason": "Near-clean UI-only: ReviewMainBody.tsx background match for the embedded review description; merges clean, zero fork drift. Batch with the next clean-port round.", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-21T18:00:13Z"} +{"sha": "2e8ff4b7", "pr": 1782, "subject": "chore: clarify shared response image guidance (#1782)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"} +{"sha": "4ea2441a", "pr": 1791, "subject": "fix: match embedded review description background (#1791)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"} {"sha": "589dfd83", "pr": 1787, "subject": "fix: Harden Stagehand browser URL handling (#1787)", "disposition": "wont-merge", "reason": "Follows #1648 (wont-merge): fork does not carry the Stagehand browser subagent — this 'fix' re-adds the entire module (992 insertions; modify/delete vs HEAD) plus the stagehand dep in pyproject/uv.lock. Adopting would resurrect a feature rejected under the curated-tools policy.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:14Z"} {"sha": "9bbd65d3", "pr": 1781, "subject": "fix: derive model context windows from LangChain profiles (#1781)", "disposition": "deferred", "reason": "Pair with deferred #1778 on context-usage-ui: derives model context windows from LangChain profiles. options.py conflicts — re-key to the fork's Bedrock/Fireworks model map (upstream IDs differ; verify LangChain profiles even resolve for the fork's Bedrock-style IDs, else keep static values). uv.lock bumps langchain to a profiles-capable version; test_model_fallback_resolution also conflicts. Port with/after #1778.", "branch": "context-usage-ui", "local_sha": null, "updated": "2026-07-21T18:00:14Z"} {"sha": "df743658", "pr": 1774, "subject": "feat: add repository skill support to the coding agent (#1774)", "disposition": "wont-merge", "reason": "Upstream reverted this feature in #1800: its factory-time ensure_sandbox_for_thread call ran outside the serialized run and raced the deterministic sandbox name (prod sandbox-create 409s spiking from the #1774 merge date). Feature withdrawn upstream; the fork's deferred security review is moot. If upstream re-lands repo skills (host-side .agents/skills resolution per #1800), triage that new commit fresh — the prompt-injection / trusted-ref concerns in the old reason still apply.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"} -{"sha": "75fb8b48", "pr": 1786, "subject": "Fix PR creation guard shell bypasses (#1786)", "disposition": "deferred", "reason": "SECURITY priority, clean merge: closes shell-bypass holes in PullRequestCreationGuardMiddleware (nested 'bash -c' expansion to depth 3, quoted-executable normalization) — a guard this fork actively wires. Port promptly; ALSO mirror the nested-shell expansion into the fork-only PullRequestVerdictGuardMiddleware (pr_verdict_guard.py), which shares the naive shlex approach and has the same bypass shape.", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-21T18:00:23Z"} -{"sha": "32e81f29", "pr": 1788, "subject": "Fix: Fix Insecure Direct Object Reference in slack_start_new_thread.py (#1788)", "disposition": "deferred", "reason": "SECURITY priority, clean merge: IDOR fix — slack_start_new_thread now enforces the deployment allowlist (_is_repo_allowed) and per-user repo access (require_repo_access_for_user keyed on the parent thread's github_login) before dispatching a run against a different repo. Both helpers exist in the fork at the same paths; merges clean. Port promptly.", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-21T18:00:23Z"} -{"sha": "ab85b372", "pr": 1764, "subject": "fix: add exc_info to swallowed exception in push re-review webhook (#1764)", "disposition": "deferred", "reason": "Trivial: adds exc_info=True to the swallowed exception log in process_github_push_event (webhooks/github.py). Merges clean. Batch with the next clean-port round.", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-21T18:00:23Z"} +{"sha": "75fb8b48", "pr": 1786, "subject": "Fix PR creation guard shell bypasses (#1786)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"} +{"sha": "32e81f29", "pr": 1788, "subject": "Fix: Fix Insecure Direct Object Reference in slack_start_new_thread.py (#1788)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"} +{"sha": "ab85b372", "pr": 1764, "subject": "fix: add exc_info to swallowed exception in push re-review webhook (#1764)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"} {"sha": "7312851d", "pr": 1777, "subject": "feat: add explicit plan approval tool (#1777)", "disposition": "deferred", "reason": "Feature: explicit approve_plan tool + plan-mode exit. All deps exist in fork (plan_store PLAN_STATUS_APPROVED/SHARED, thread_api._user_owns_thread). Conflicts: prompt.py (fork prompt constants), server.py (tool/middleware wiring), check_message_queue.py (fork dashboard-handoff + mid-run injection drift), AGENTS.md. Hand re-key those four; keep the whole vertical (tool + plan_mode + thread_api + 4 test files) on one side.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:33Z"} {"sha": "e51abe14", "pr": 1754, "subject": "fix: skip oversized images before model calls (#1754)", "disposition": "deferred", "reason": "Desirable: 10MB image size cap + 'image omitted' text block before model calls, prevents oversized-image model failures. Conflicts only because the fork hardened fetch_image_block (SSRF redirect guard, host-only logging) — hunks are compatible; re-key the size check around the fork's guarded fetch and port impl + test together.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:33Z"} {"sha": "0ed560a5", "pr": 1779, "subject": "fix: settle review checks after run failures (#1779)", "disposition": "deferred", "reason": "Desirable: run-failure completion webhook now settles reviewer check runs left open when the graph dies (_settle_failed_reviewer_check; uses settle_review_check_run + review_check_pending_result, both already in the fork's review/publish.py from the #214/#215 ports). Conflict is fork drift in completion.py (failure-reply customizations); re-key the new helper in and port with its 159-line test.", "branch": "", "local_sha": null, "updated": "2026-07-21T18:00:33Z"} @@ -146,7 +146,7 @@ {"sha": "bedc57b8", "pr": 1796, "subject": "fix: cap execute output by making SandboxBackendProxy a BaseSandbox (#1796)", "disposition": "deferred", "reason": "Real fix (unbounded execute stdout pulled into the worker → OOM risk) but inapplicable at the fork's deepagents==0.6.12: no capture-offload API exists (ExecuteOffloadResult / execute_accepts_timeout absent; FilesystemMiddleware has no _resolve_capture BaseSandbox gate), so the bug it fixes cannot occur yet. Re-triage together with deferred #1745 (deepagents 0.6.12→0.7.x bump) — landing that bump makes this fix required. Fork's SandboxBackendProxy has diverged (sync-era, bound_repo repo-binding, no reconnect machinery): hand-apply the execute_with_offload/aexecute_with_offload delegation onto the fork proxy rather than cherry-pick.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"} {"sha": "e1138cf5", "pr": 1797, "subject": "fix: merge concurrent trusted skill refs (#1797)", "disposition": "wont-merge", "reason": "Follow-up fix to #1774's TrustedSkillsMiddleware, which the fork never adopted (row df743658) and upstream itself reverted in #1800 — nothing to apply.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"} {"sha": "60e7307c", "pr": 1800, "subject": "revert: repository skill support for the coding agent (#1774, #1797) (#1800)", "disposition": "wont-merge", "reason": "Revert of #1774 + #1797; the fork adopted neither, so there is nothing to revert. Upstream's rationale: #1774's factory-time ensure_sandbox_for_thread ran outside the serialized run and raced the thread-deterministic sandbox name (prod create-409s). Fork invariant holds: its single provisioning call sits inside the __creating__-sentinel lifecycle within the interrupt-serialized run.", "branch": "", "local_sha": null, "updated": "2026-07-23T16:54:54Z"} -{"sha": "a77c4e47", "pr": 1799, "subject": "fix: show current shared thread in sidebar (#1799)", "disposition": "deferred", "reason": "Desirable dashboard fix: sidebar now surfaces the currently-open shared thread (org-member 'Open in Web' links) when it falls outside the caller's own list — _sidebar_active_thread_summary + active_thread_id param; fork has the same include_all/org-member read path in thread_api.py. Cherry-picks CLEAN onto dev (merge-tree probe: zero conflicts); port with its thread-api tests + sandbox_id e2e spec + AgentsSidebar/queries UI as one vertical.", "branch": "", "local_sha": null, "updated": "2026-07-24T19:25:32Z"} +{"sha": "a77c4e47", "pr": 1799, "subject": "fix: show current shared thread in sidebar (#1799)", "disposition": "landed", "reason": "Landed via fork PR #226", "branch": "feature/upstream-clean-batch-jul21", "local_sha": null, "updated": "2026-07-24T19:38:23Z"} {"sha": "b67215ed", "pr": 1780, "subject": "feat: link originating Slack threads (#1780)", "disposition": "deferred", "reason": "Desirable for the Slack-heavy fork: dispatch records the originating Slack permalink in thread metadata (webhooks/common.py), open_pull_request embeds an 'Originating Slack thread' link in the PR body (gated to private repos), and the sidebar menu links it. Impl files auto-merge (thread_api / open_pull_request / webhooks/common); conflicts are fork drift in prompt.py (customized prompt constants), tests/e2e/harness.py, tests/slack/test_slack_context.py — hand re-key those; keep backend + UI (AgentsSidebar, types) + e2e dashboard spec as one vertical.", "branch": "", "local_sha": null, "updated": "2026-07-24T19:25:37Z"} {"sha": "7d79d300", "pr": 1804, "subject": "chore(deps): bump deepagents to 0.7.0b1 pin (#1804)", "disposition": "deferred", "reason": "Rides the deferred #1745 deepagents bump: 0.7.0a7→0.7.0b1 pin + wcmatch>=11.0 uv override (langchain-e2b's e2b dep still caps wcmatch<11.0). Fork is pinned deepagents==0.6.12 — and langsmith==0.10.10 already satisfies the new >=0.10.9 floor — and never took the a7-era HARNESS_EXCLUDED_MIDDLEWARE code its prompt.py hunk deletes, so nothing applies standalone. Re-triage as one cluster: #1745 (bump, now targeting 0.7.0b1) + #1796 (offload delegation) + this pin/override.", "branch": "", "local_sha": null, "updated": "2026-07-24T19:25:44Z"} {"sha": "0e2eefbc", "pr": 1801, "subject": "chore: bump default sandbox resources and delete-after-stop (#1801)", "disposition": "deferred", "reason": "Capacity/cost policy, not a correctness fix: defaults 2→4 vCPU, 7936MiB→16GiB mem, 32→128GiB FS, stopped-sandbox delete 24h→14d. Would materially raise LangSmith sandbox spend on sh-openswe unless prod env already overrides; all five values are env-overridable today. Decide prod sizing first (ops decision), then adopt or reject. Textual conflict is only fork drift in integrations/langsmith.py (sync-era SandboxClient, reconnect constants absent) — the constants re-key trivially; docs hunks auto-merge.", "branch": "", "local_sha": null, "updated": "2026-07-24T19:25:49Z"} diff --git a/docs/upstream-sync/triage.md b/docs/upstream-sync/triage.md index 9c2cf066..44c97697 100644 --- a/docs/upstream-sync/triage.md +++ b/docs/upstream-sync/triage.md @@ -88,6 +88,12 @@ Rows key on the **upstream SHA** (stable across local cherry-picks). Deferred ro | `b5e52925` | #1775 | feat: add structured Linear issue filters (#1775) | Landed | Clean pick: structured filters for linear_search_issues — stacks directly on #1748 (landed PR #206); zero fork drift on all three files. Ready whenever. | feature/upstream-clean-batch-security-linear | | `31263f83` | #1785 | Fix: Fix Stored XSS in ReplyCard.tsx (#1785) | Landed | SECURITY priority, near-clean: diff is urlencode() hardening of the GitHub OAuth authorize URL in dashboard routes.py auth_login (upstream bot title says ReplyCard.tsx — mismatch, trust the diff). Fork auth_login has the identical f-string URL; hunk applies clean. Port promptly. | feature/upstream-clean-batch-security-linear | | `3ea29d3f` | #1789 | Fix: Fix Improper privilege management in server.py (#1789) | Landed | SECURITY priority, near-clean: adds empty-signature reject to verify_github_signature (utils/github_comments.py) + verify_linear_signature (webhooks/common.py) (title says server.py — mismatch, trust the diff). Both fork functions match at the hunk sites; fork-only verify_jira_secret already guards empty token. Real value: None signature currently raises TypeError in compare_digest. Port BOTH hunks together. | feature/upstream-clean-batch-security-linear | +| `2e8ff4b7` | #1782 | chore: clarify shared response image guidance (#1782) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 | +| `4ea2441a` | #1791 | fix: match embedded review description background (#1791) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 | +| `75fb8b48` | #1786 | Fix PR creation guard shell bypasses (#1786) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 | +| `32e81f29` | #1788 | Fix: Fix Insecure Direct Object Reference in slack_start_new_thread.py (#1788) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 | +| `ab85b372` | #1764 | fix: add exc_info to swallowed exception in push re-review webhook (#1764) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 | +| `a77c4e47` | #1799 | fix: show current shared thread in sidebar (#1799) | Landed | Landed via fork PR #226 | feature/upstream-clean-batch-jul21 | | `c3292d82` | #1611 | bake sfw binary into sandbox image | Won't merge | already in dev | | | `48bf712b` | #1609 | show message timestamps | Won't merge | already in dev | | | `85c0f63e` | #1620 | clickable shared PR header | Won't merge | already in dev | | @@ -147,17 +153,11 @@ Rows key on the **upstream SHA** (stable across local cherry-picks). Deferred ro | `81d544bc` | #1765 | feat: Expose more specific AGENTS.md context to Open SWE reviewer (#1765) | Deferred | Near-clean: agents_md.py helper + tests are zero-drift; reviewer.py hunk is small (~39 lines) but lands in the fork's heavily-diverged reviewer — hand-apply the scoped_agents_md wiring onto the fork's fetch_agents_md call sites (reviewer.py ~L404/445/1031). | reviewer-context | | `8c8e58bc` | #1776 | feat: connect automations to Slack channels (#1776) | Deferred | Moderate reconcile: Slack-channel wiring for automations. Fork helpers exist (post_slack_top_level_message_with_ts, generate_thread_id_from_slack_thread, slack_id_for_login); completion.py (+233 drift) and schedules.py (+167) need hand-merge; UI automations feature present. Keep backend+UI+tests as one vertical. | automations-slack | | `f0897479` | #1778 | feat: surface context window usage in agents UI (#1778) | Deferred | Near-clean: context-window indicator UI is all new files (zero drift); options.py hunk must be re-keyed to the fork's Bedrock/Fireworks model map (fork model IDs differ from upstream's) — add context_window per fork entry rather than taking upstream values. | context-usage-ui | -| `2e8ff4b7` | #1782 | chore: clarify shared response image guidance (#1782) | Deferred | Near-clean: docstring-only guidance in save_plan (shared responses persist Markdown only, never sandbox-local images — post screenshots directly to Slack) + 2 assertions in test_plan_review. Merges clean against fork. Batch with the next clean-port round. | feature/upstream-clean-batch-jul21 | -| `4ea2441a` | #1791 | fix: match embedded review description background (#1791) | Deferred | Near-clean UI-only: ReviewMainBody.tsx background match for the embedded review description; merges clean, zero fork drift. Batch with the next clean-port round. | feature/upstream-clean-batch-jul21 | | `9bbd65d3` | #1781 | fix: derive model context windows from LangChain profiles (#1781) | Deferred | Pair with deferred #1778 on context-usage-ui: derives model context windows from LangChain profiles. options.py conflicts — re-key to the fork's Bedrock/Fireworks model map (upstream IDs differ; verify LangChain profiles even resolve for the fork's Bedrock-style IDs, else keep static values). uv.lock bumps langchain to a profiles-capable version; test_model_fallback_resolution also conflicts. Port with/after #1778. | context-usage-ui | -| `75fb8b48` | #1786 | Fix PR creation guard shell bypasses (#1786) | Deferred | SECURITY priority, clean merge: closes shell-bypass holes in PullRequestCreationGuardMiddleware (nested 'bash -c' expansion to depth 3, quoted-executable normalization) — a guard this fork actively wires. Port promptly; ALSO mirror the nested-shell expansion into the fork-only PullRequestVerdictGuardMiddleware (pr_verdict_guard.py), which shares the naive shlex approach and has the same bypass shape. | feature/upstream-clean-batch-jul21 | -| `32e81f29` | #1788 | Fix: Fix Insecure Direct Object Reference in slack_start_new_thread.py (#1788) | Deferred | SECURITY priority, clean merge: IDOR fix — slack_start_new_thread now enforces the deployment allowlist (_is_repo_allowed) and per-user repo access (require_repo_access_for_user keyed on the parent thread's github_login) before dispatching a run against a different repo. Both helpers exist in the fork at the same paths; merges clean. Port promptly. | feature/upstream-clean-batch-jul21 | -| `ab85b372` | #1764 | fix: add exc_info to swallowed exception in push re-review webhook (#1764) | Deferred | Trivial: adds exc_info=True to the swallowed exception log in process_github_push_event (webhooks/github.py). Merges clean. Batch with the next clean-port round. | feature/upstream-clean-batch-jul21 | | `7312851d` | #1777 | feat: add explicit plan approval tool (#1777) | Deferred | Feature: explicit approve_plan tool + plan-mode exit. All deps exist in fork (plan_store PLAN_STATUS_APPROVED/SHARED, thread_api._user_owns_thread). Conflicts: prompt.py (fork prompt constants), server.py (tool/middleware wiring), check_message_queue.py (fork dashboard-handoff + mid-run injection drift), AGENTS.md. Hand re-key those four; keep the whole vertical (tool + plan_mode + thread_api + 4 test files) on one side. | | | `e51abe14` | #1754 | fix: skip oversized images before model calls (#1754) | Deferred | Desirable: 10MB image size cap + 'image omitted' text block before model calls, prevents oversized-image model failures. Conflicts only because the fork hardened fetch_image_block (SSRF redirect guard, host-only logging) — hunks are compatible; re-key the size check around the fork's guarded fetch and port impl + test together. | | | `0ed560a5` | #1779 | fix: settle review checks after run failures (#1779) | Deferred | Desirable: run-failure completion webhook now settles reviewer check runs left open when the graph dies (_settle_failed_reviewer_check; uses settle_review_check_run + review_check_pending_result, both already in the fork's review/publish.py from the #214/#215 ports). Conflict is fork drift in completion.py (failure-reply customizations); re-key the new helper in and port with its 159-line test. | | | `bedc57b8` | #1796 | fix: cap execute output by making SandboxBackendProxy a BaseSandbox (#1796) | Deferred | Real fix (unbounded execute stdout pulled into the worker → OOM risk) but inapplicable at the fork's deepagents==0.6.12: no capture-offload API exists (ExecuteOffloadResult / execute_accepts_timeout absent; FilesystemMiddleware has no _resolve_capture BaseSandbox gate), so the bug it fixes cannot occur yet. Re-triage together with deferred #1745 (deepagents 0.6.12→0.7.x bump) — landing that bump makes this fix required. Fork's SandboxBackendProxy has diverged (sync-era, bound_repo repo-binding, no reconnect machinery): hand-apply the execute_with_offload/aexecute_with_offload delegation onto the fork proxy rather than cherry-pick. | | -| `a77c4e47` | #1799 | fix: show current shared thread in sidebar (#1799) | Deferred | Desirable dashboard fix: sidebar now surfaces the currently-open shared thread (org-member 'Open in Web' links) when it falls outside the caller's own list — _sidebar_active_thread_summary + active_thread_id param; fork has the same include_all/org-member read path in thread_api.py. Cherry-picks CLEAN onto dev (merge-tree probe: zero conflicts); port with its thread-api tests + sandbox_id e2e spec + AgentsSidebar/queries UI as one vertical. | | | `b67215ed` | #1780 | feat: link originating Slack threads (#1780) | Deferred | Desirable for the Slack-heavy fork: dispatch records the originating Slack permalink in thread metadata (webhooks/common.py), open_pull_request embeds an 'Originating Slack thread' link in the PR body (gated to private repos), and the sidebar menu links it. Impl files auto-merge (thread_api / open_pull_request / webhooks/common); conflicts are fork drift in prompt.py (customized prompt constants), tests/e2e/harness.py, tests/slack/test_slack_context.py — hand re-key those; keep backend + UI (AgentsSidebar, types) + e2e dashboard spec as one vertical. | | | `7d79d300` | #1804 | chore(deps): bump deepagents to 0.7.0b1 pin (#1804) | Deferred | Rides the deferred #1745 deepagents bump: 0.7.0a7→0.7.0b1 pin + wcmatch>=11.0 uv override (langchain-e2b's e2b dep still caps wcmatch<11.0). Fork is pinned deepagents==0.6.12 — and langsmith==0.10.10 already satisfies the new >=0.10.9 floor — and never took the a7-era HARNESS_EXCLUDED_MIDDLEWARE code its prompt.py hunk deletes, so nothing applies standalone. Re-triage as one cluster: #1745 (bump, now targeting 0.7.0b1) + #1796 (offload delegation) + this pin/override. | | | `0e2eefbc` | #1801 | chore: bump default sandbox resources and delete-after-stop (#1801) | Deferred | Capacity/cost policy, not a correctness fix: defaults 2→4 vCPU, 7936MiB→16GiB mem, 32→128GiB FS, stopped-sandbox delete 24h→14d. Would materially raise LangSmith sandbox spend on sh-openswe unless prod env already overrides; all five values are env-overridable today. Decide prod sizing first (ops decision), then adopt or reject. Textual conflict is only fork drift in integrations/langsmith.py (sync-era SandboxClient, reconnect constants absent) — the constants re-key trivially; docs hunks auto-merge. | | diff --git a/tests/agent/test_plan_review.py b/tests/agent/test_plan_review.py index d6c321d4..7cc49219 100644 --- a/tests/agent/test_plan_review.py +++ b/tests/agent/test_plan_review.py @@ -214,6 +214,14 @@ def test_save_plan_exported_and_wired() -> None: assert callable(save_plan) +def test_save_plan_description_warns_about_slack_images() -> None: + from agent.tools import save_plan + + description = save_plan.__doc__ or "" + assert "persist Markdown text only" in description + assert "post them directly in Slack" in description + + def test_plan_status_constants() -> None: from agent.dashboard import plan_store diff --git a/tests/dashboard/test_dashboard_thread_api.py b/tests/dashboard/test_dashboard_thread_api.py index d409f291..7cad7a99 100644 --- a/tests/dashboard/test_dashboard_thread_api.py +++ b/tests/dashboard/test_dashboard_thread_api.py @@ -1360,6 +1360,161 @@ async def test_list_dashboard_threads_sidebar_fills_buckets_with_one_endpoint(mo assert {call["offset"] for call in searches} == {0, page_size} +async def test_list_dashboard_threads_sidebar_includes_readable_active_thread( + monkeypatch, +) -> None: + threads = _make_threads(1, resolved_before=0) + shared_thread = { + "thread_id": "shared-thread", + "metadata": { + "source": "slack", + "github_login": "teammate", + "title": "Teammate thread", + "updated_at_ms": 100, + "latest_run_status": "success", + "sandbox_id": "sandbox-123", + }, + } + + class FakeThreads: + async def search(self, *, metadata, limit, offset, sort_by, sort_order, select): + assert select == thread_api._THREAD_LIST_SELECT + return threads[offset : offset + limit] + + async def get(self, thread_id): + assert thread_id == "shared-thread" + return shared_thread + + async def update(self, *, thread_id, metadata): + return None + + class FakeRuns: + async def list(self, thread_id, limit=1): + return [] + + class FakeClient: + threads = FakeThreads() + runs = FakeRuns() + + monkeypatch.setattr(thread_api, "langgraph_client", lambda: FakeClient()) + + result = await thread_api.list_dashboard_threads_sidebar( + "octocat", + email=None, + active_limit=5, + resolved_limit=5, + active_thread_id="shared-thread", + ) + + assert [item["id"] for item in result["active"]["items"]] == ["shared-thread", "t0"] + shared = result["active"]["items"][0] + assert shared["isOwner"] is False + assert shared["sandboxId"] == "sandbox-123" + + +async def test_list_dashboard_threads_sidebar_keeps_resolved_active_thread_resolved( + monkeypatch, +) -> None: + threads = _make_threads(1, resolved_before=0) + shared_thread = { + "thread_id": "shared-resolved-thread", + "metadata": { + "source": "slack", + "github_login": "teammate", + "title": "Resolved teammate thread", + "updated_at_ms": 100, + "latest_run_status": "success", + "resolved": True, + "sandbox_id": "sandbox-456", + }, + } + + class FakeThreads: + async def search(self, *, metadata, limit, offset, sort_by, sort_order, select): + assert select == thread_api._THREAD_LIST_SELECT + return threads[offset : offset + limit] + + async def get(self, thread_id): + assert thread_id == "shared-resolved-thread" + return shared_thread + + async def update(self, *, thread_id, metadata): + return None + + class FakeRuns: + async def list(self, thread_id, limit=1): + return [] + + class FakeClient: + threads = FakeThreads() + runs = FakeRuns() + + monkeypatch.setattr(thread_api, "langgraph_client", lambda: FakeClient()) + + result = await thread_api.list_dashboard_threads_sidebar( + "octocat", + email=None, + active_limit=5, + resolved_limit=5, + active_thread_id="shared-resolved-thread", + ) + + assert [item["id"] for item in result["active"]["items"]] == ["t0"] + assert [item["id"] for item in result["resolved"]["items"]] == ["shared-resolved-thread"] + shared = result["resolved"]["items"][0] + assert shared["isOwner"] is False + assert shared["resolved"] is True + assert shared["sandboxId"] == "sandbox-456" + + +async def test_list_dashboard_threads_sidebar_ignores_unreadable_active_thread( + monkeypatch, +) -> None: + threads = _make_threads(1, resolved_before=0) + private_thread = { + "thread_id": "private-thread", + "metadata": { + "source": "internal", + "github_login": "teammate", + "title": "Private thread", + "updated_at_ms": 100, + "latest_run_status": "success", + }, + } + + class FakeThreads: + async def search(self, *, metadata, limit, offset, sort_by, sort_order, select): + assert select == thread_api._THREAD_LIST_SELECT + return threads[offset : offset + limit] + + async def get(self, thread_id): + assert thread_id == "private-thread" + return private_thread + + async def update(self, *, thread_id, metadata): + return None + + class FakeRuns: + async def list(self, thread_id, limit=1): + return [] + + class FakeClient: + threads = FakeThreads() + runs = FakeRuns() + + monkeypatch.setattr(thread_api, "langgraph_client", lambda: FakeClient()) + + result = await thread_api.list_dashboard_threads_sidebar( + "octocat", + email=None, + active_limit=5, + resolved_limit=5, + active_thread_id="private-thread", + ) + + assert [item["id"] for item in result["active"]["items"]] == ["t0"] + + async def test_list_dashboard_threads_page_refreshes_only_unsettled_threads(monkeypatch) -> None: threads = _make_threads(3, resolved_before=0) threads[0]["metadata"]["latest_run_status"] = "success" diff --git a/tests/e2e/tests/sandbox_id.spec.ts b/tests/e2e/tests/sandbox_id.spec.ts index 8d21f7a2..3c6ed3c0 100644 --- a/tests/e2e/tests/sandbox_id.spec.ts +++ b/tests/e2e/tests/sandbox_id.spec.ts @@ -5,6 +5,7 @@ import { test, expect, type Locator, type Page } from "@playwright/test"; // creates a sandbox and stamps its id into the thread metadata, and the UI // copies that same id. Only the LLM/GitHub/Slack boundaries are faked. const SAME_USER = { login: "alice", email: "alice@example.com" }; +const OTHER_USER = { login: "bob", email: "bob@example.com" }; async function loginAs(page: Page, user: { login: string; email: string }) { const res = await page.request.post("/control/login", { data: user }); @@ -112,4 +113,39 @@ test.describe("thread sandbox id (real dashboard UI)", () => { await context.close(); }); + + test("shared active thread appears in the sidebar with sandbox action", async ({ + page, + browser, + baseURL, + }, testInfo) => { + await loginAs(page, SAME_USER); + const threadId = await createThreadWithSandbox(page); + + const bobContext = await browser.newContext({ baseURL }); + await bobContext.grantPermissions(["clipboard-read", "clipboard-write"], { + origin: baseURL, + }); + const bobPage = await bobContext.newPage(); + await loginAs(bobPage, OTHER_USER); + await bobPage.goto(`/agents/${threadId}`); + await expect(bobPage).toHaveURL(new RegExp(`/agents/${threadId}$`)); + + const row = bobPage.locator(`a[href$="/agents/${threadId}"]`).first(); + await expect(row).toBeVisible(); + await row.hover(); + await kebabFor(row).click(); + await expect(copyItem(bobPage)).toBeEnabled(); + + const screenshotPath = testInfo.outputPath( + "shared-thread-sidebar-sandbox-menu.png", + ); + await bobPage.screenshot({ path: screenshotPath, fullPage: true }); + await testInfo.attach("shared-thread-sidebar-sandbox-menu", { + path: screenshotPath, + contentType: "image/png", + }); + + await bobContext.close(); + }); }); diff --git a/tests/github/test_pr_creation_guard.py b/tests/github/test_pr_creation_guard.py index 55b9060d..a993bd41 100644 --- a/tests/github/test_pr_creation_guard.py +++ b/tests/github/test_pr_creation_guard.py @@ -38,6 +38,23 @@ def test_detects_pr_creation_fallback_commands() -> None: assert is_pr_creation_fallback_command( "curl -X POST https://api.github.com/repos/langchain-ai/open-swe/pulls -d '{}'" ) + assert is_pr_creation_fallback_command("/usr/bin/gh pr create --draft") + assert is_pr_creation_fallback_command( + "/usr/bin/curl -X POST https://api.github.com/repos/langchain-ai/open-swe/pulls -d '{}'" + ) + assert is_pr_creation_fallback_command("bash -c 'gh pr create --draft'") + assert is_pr_creation_fallback_command("bash -c'gh pr create --draft'") + assert is_pr_creation_fallback_command("zsh -lc'gh pr create --draft'") + assert is_pr_creation_fallback_command("GH_TOKEN=dummy sh -c 'gh pr create --draft'") + assert is_pr_creation_fallback_command( + "zsh -lc 'gh api repos/langchain-ai/open-swe/pulls -X POST -f title=x'" + ) + assert is_pr_creation_fallback_command( + "bash -c \"curl -X POST https://api.github.com/repos/langchain-ai/open-swe/pulls -d '{}'\"" + ) + assert is_pr_creation_fallback_command( + 'bash -c \'sh -c "dash -c \\"zsh -c \\\\\\"gh pr create --draft\\\\\\"\\""\'' + ) def test_allows_safe_pr_commands() -> None: @@ -45,6 +62,8 @@ def test_allows_safe_pr_commands() -> None: assert not is_pr_creation_fallback_command("gh pr list --head open-swe/foo") assert not is_pr_creation_fallback_command("gh pr edit 1 --add-label ready") assert not is_pr_creation_fallback_command("gh pr comment 1 --body done") + assert not is_pr_creation_fallback_command("bash -c 'gh pr view 1 --json url'") + assert not is_pr_creation_fallback_command("/usr/bin/gh pr view 1 --json url") async def test_middleware_blocks_execute_pr_creation_fallbacks() -> None: diff --git a/tests/github/test_pr_verdict_guard.py b/tests/github/test_pr_verdict_guard.py index d86908df..c3c1fd81 100644 --- a/tests/github/test_pr_verdict_guard.py +++ b/tests/github/test_pr_verdict_guard.py @@ -60,6 +60,28 @@ def test_detects_curl_review_verdicts() -> None: ) +def test_detects_nested_and_normalized_verdict_commands() -> None: + assert is_pr_verdict_fallback_command("/usr/bin/gh pr review 12 --approve") + assert is_pr_verdict_fallback_command( + "/usr/bin/curl -X POST https://api.github.com/repos/o/r/pulls/5/reviews " + '-d \'{"event": "APPROVE"}\'' + ) + assert is_pr_verdict_fallback_command("bash -c 'gh pr review 12 --approve'") + assert is_pr_verdict_fallback_command("bash -c'gh pr review 12 --approve'") + assert is_pr_verdict_fallback_command("zsh -lc'gh pr review 12 --approve'") + assert is_pr_verdict_fallback_command("GH_TOKEN=dummy sh -c 'gh pr review -a'") + assert is_pr_verdict_fallback_command( + "zsh -lc 'gh api repos/o/r/pulls/5/reviews -X POST -f event=REQUEST_CHANGES'" + ) + assert is_pr_verdict_fallback_command( + 'bash -c "curl -X POST https://api.github.com/repos/o/r/pulls/5/reviews ' + '-d \'{\\"event\\": \\"APPROVE\\"}\'"' + ) + assert is_pr_verdict_fallback_command( + 'bash -c \'sh -c "dash -c \\"zsh -c \\\\\\"gh pr review 12 --approve\\\\\\"\\""\'' + ) + + def test_allows_safe_review_commands() -> None: assert not is_pr_verdict_fallback_command("gh pr review 12 --comment -b 'looks good'") assert not is_pr_verdict_fallback_command("gh pr review -c") @@ -69,6 +91,8 @@ def test_allows_safe_review_commands() -> None: assert not is_pr_verdict_fallback_command("gh api repos/o/r/pulls/5/reviews") assert not is_pr_verdict_fallback_command("gh pr create --draft") assert not is_pr_verdict_fallback_command("curl https://api.github.com/repos/o/r/pulls/5") + assert not is_pr_verdict_fallback_command("bash -c 'gh pr review 12 --comment -b ok'") + assert not is_pr_verdict_fallback_command("/usr/bin/gh pr view 12 --json reviews") async def test_middleware_blocks_execute_verdict_fallbacks() -> None: diff --git a/ui/src/features/agents/components/AgentsSidebar.tsx b/ui/src/features/agents/components/AgentsSidebar.tsx index b6057ac5..63caf9b8 100644 --- a/ui/src/features/agents/components/AgentsSidebar.tsx +++ b/ui/src/features/agents/components/AgentsSidebar.tsx @@ -111,7 +111,7 @@ export function AgentsSidebar({ }: AgentsSidebarProps) { const { prefs, setGroup, setCompact, setFilters, resetFilters } = useSidebarPrefs() - const sidebar = useSidebarThreads(RESOLVED_SIDEBAR_LIMIT) + const sidebar = useSidebarThreads(RESOLVED_SIDEBAR_LIMIT, activeThreadId) const activeThreads = sidebar.data?.active.items ?? [] const resolvedThreads = sidebar.data?.resolved.items ?? [] const resolvedHasMore = sidebar.data?.resolved.hasMore ?? false diff --git a/ui/src/features/agents/lib/api.ts b/ui/src/features/agents/lib/api.ts index bb0dc0f6..42d84acd 100644 --- a/ui/src/features/agents/lib/api.ts +++ b/ui/src/features/agents/lib/api.ts @@ -225,6 +225,7 @@ function buildThreadsPageQuery(params: ThreadsPageParams): string { function buildSidebarThreadsQuery(params: { activeLimit?: number resolvedLimit?: number + activeThreadId?: string }): string { const search = new URLSearchParams() if (params.activeLimit != null) { @@ -233,6 +234,9 @@ function buildSidebarThreadsQuery(params: { if (params.resolvedLimit != null) { search.set("resolved_limit", String(params.resolvedLimit)) } + if (params.activeThreadId) { + search.set("active_thread_id", params.activeThreadId) + } const query = search.toString() return query ? `?${query}` : "" } @@ -242,6 +246,7 @@ export const agentsApi = { listSidebarThreads: (params: { activeLimit?: number resolvedLimit?: number + activeThreadId?: string }) => agentsRequest( `/threads/sidebar${buildSidebarThreadsQuery(params)}` diff --git a/ui/src/features/agents/lib/queries.ts b/ui/src/features/agents/lib/queries.ts index a5a50202..925fc1b0 100644 --- a/ui/src/features/agents/lib/queries.ts +++ b/ui/src/features/agents/lib/queries.ts @@ -13,8 +13,11 @@ import type { AgentThread, Chunk, ImageChunk, Message } from "./types" export const agentThreadKeys = { lists: ["agent-threads", "lists"] as const, - sidebar: (params: { activeLimit: number; resolvedLimit: number }) => - ["agent-threads", "lists", "sidebar", params] as const, + sidebar: (params: { + activeLimit: number + resolvedLimit: number + activeThreadId?: string + }) => ["agent-threads", "lists", "sidebar", params] as const, detail: (threadId: string) => ["agent-threads", threadId] as const, prDiff: (threadId: string) => ["agent-threads", threadId, "pr-diff"] as const, workflowApprovals: (threadId: string) => @@ -93,10 +96,14 @@ function sidebarRefetchInterval(query: { state: { data?: SidebarThreads } }) { : false } -export function useSidebarThreads(resolvedLimit: number) { +export function useSidebarThreads( + resolvedLimit: number, + activeThreadId?: string +) { const params = { activeLimit: SIDEBAR_ACTIVE_LIMIT, resolvedLimit, + activeThreadId, } return useQuery({ queryKey: agentThreadKeys.sidebar(params), diff --git a/ui/src/features/reviews/components/ReviewMainBody.tsx b/ui/src/features/reviews/components/ReviewMainBody.tsx index 9322a334..60652602 100644 --- a/ui/src/features/reviews/components/ReviewMainBody.tsx +++ b/ui/src/features/reviews/components/ReviewMainBody.tsx @@ -1262,7 +1262,12 @@ function ReviewBodyInner({ deletions: detail.pr.deletions, }} /> -
+
{detail.pr.body ? (