From a82da1f90761e75d72451c64ef5bb375d5f23235 Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Fri, 17 Jul 2026 13:45:39 -0400 Subject: [PATCH 1/7] refactor: move docs/resources/assets to domain layout Part of the domain-reorg adoption (build plan step C1): fork content, upstream layout. Moves INSTALLATION.md/CUSTOMIZATION.md under docs/, static/ under assets/, and default_prompt.md under agent/resources/ (packaged via agent/resources/__init__.py), then switches prompt.py's loader to importlib.resources with an explicit DEFAULT_PROMPT_PATH override, matching upstream's hunk. README and CUSTOMIZATION.md links updated for the new paths; wheel build verified to still ship agent/resources/default_prompt.md. --- README.md | 18 +++++------ agent/prompt.py | 31 ++++++++++++------- agent/resources/__init__.py | 1 + .../resources/default_prompt.md | 0 {static => assets}/dark.svg | 0 {static => assets}/light.svg | 0 CUSTOMIZATION.md => docs/CUSTOMIZATION.md | 2 +- INSTALLATION.md => docs/INSTALLATION.md | 0 8 files changed, 30 insertions(+), 22 deletions(-) create mode 100644 agent/resources/__init__.py rename default_prompt.md => agent/resources/default_prompt.md (100%) rename {static => assets}/dark.svg (100%) rename {static => assets}/light.svg (100%) rename CUSTOMIZATION.md => docs/CUSTOMIZATION.md (99%) rename INSTALLATION.md => docs/INSTALLATION.md (100%) diff --git a/README.md b/README.md index 4e34f72d..8232d10c 100644 --- a/README.md +++ b/README.md @@ -1,9 +1,9 @@
- - - Open SWE Logo + + + Open SWE Logo
@@ -54,7 +54,7 @@ create_deep_agent( Every task runs in its own **isolated cloud sandbox** — a remote Linux environment with full shell access. The repo is cloned in, the agent gets full permissions, and the blast radius of any mistake is fully contained. No production access, no confirmation prompts. -Open SWE supports multiple sandbox providers out of the box — [Modal](https://modal.com/), [Daytona](https://www.daytona.io/), [Runloop](https://www.runloop.ai/), [E2B](https://e2b.dev/), and [LangSmith](https://smith.langchain.com/) — and you can plug in your own. See the [Customization Guide](CUSTOMIZATION.md#1-sandbox) for details. +Open SWE supports multiple sandbox providers out of the box — [Modal](https://modal.com/), [Daytona](https://www.daytona.io/), [Runloop](https://www.runloop.ai/), [E2B](https://e2b.dev/), and [LangSmith](https://smith.langchain.com/) — and you can plug in your own. See the [Customization Guide](docs/CUSTOMIZATION.md#1-sandbox) for details. This follows the principle all three companies converge on: **isolate first, then give full permissions inside the boundary.** @@ -112,7 +112,7 @@ All three companies in the article converge on **Slack as the primary invocation - **Confluence** — Comment `@openswe` on a page. A private Atlassian Connect app delivers the `comment_created` event; the agent acts and replies on the page. - **GitHub** — Tag `@openswe` in PR comments on agent-created PRs to have it address review feedback and push fixes to the same branch. -See **[INSTALLATION.md](./INSTALLATION.md) §5** for per-surface trigger setup. +See **[INSTALLATION.md](./docs/INSTALLATION.md) §5** for per-surface trigger setup. Each invocation creates a deterministic thread ID, so follow-up messages on the same issue or thread route to the same running agent. @@ -123,7 +123,7 @@ Each invocation creates a deterministic thread ID, so follow-up messages on the ### 7. Validation — Prompt-Driven The agent is instructed to run linters, formatters, and tests before committing, and is responsible end-to-end for committing, pushing, opening/updating the draft PR, and replying in the source channel. -This is an area where you can extend Open SWE for your org: add deterministic CI checks, visual verification, or review gates as additional middleware. See the [Customization Guide](CUSTOMIZATION.md#6-middleware) for how. +This is an area where you can extend Open SWE for your org: add deterministic CI checks, visual verification, or review gates as additional middleware. See the [Customization Guide](docs/CUSTOMIZATION.md#6-middleware) for how. --- @@ -156,8 +156,8 @@ This is an area where you can extend Open SWE for your org: add deterministic CI ## Getting Started -- **[Installation Guide](INSTALLATION.md)** — local dev (backend + dashboard), GitHub App creation, LangSmith, Linear/Slack/GitHub triggers, and production deployment -- **[Customization Guide](CUSTOMIZATION.md)** — swap the sandbox, model, tools, triggers, system prompt, and middleware for your org +- **[Installation Guide](docs/INSTALLATION.md)** — local dev (backend + dashboard), GitHub App creation, LangSmith, Linear/Slack/GitHub triggers, and production deployment +- **[Customization Guide](docs/CUSTOMIZATION.md)** — swap the sandbox, model, tools, triggers, system prompt, and middleware for your org ## Deployment (Sea Haven fork) @@ -169,7 +169,7 @@ and secrets live in the LangGraph deployment config and Vercel environment variables. Promotion from `dev` to `prod` (`main`) is handled by [`.github/workflows/promote-to-main.yml`](.github/workflows/promote-to-main.yml). -See **[INSTALLATION.md § 10 "Production deployment"](INSTALLATION.md#10-production-deployment)** +See **[INSTALLATION.md § 10 "Production deployment"](docs/INSTALLATION.md#10-production-deployment)** for the full backend + dashboard setup. > The earlier self-hosted AWS stack (CDK under `infra/`, an ARM64 EC2 box + nginx diff --git a/agent/prompt.py b/agent/prompt.py index 2af21ed8..fee871c1 100644 --- a/agent/prompt.py +++ b/agent/prompt.py @@ -1,6 +1,7 @@ import logging import os import shlex +from importlib import resources from pathlib import Path from deepagents import HarnessProfile, register_harness_profile @@ -14,10 +15,7 @@ from .utils.github_comments import UNTRUSTED_GITHUB_COMMENT_OPEN_TAG logger = logging.getLogger(__name__) -DEFAULT_PROMPT_PATH = os.environ.get( - "DEFAULT_PROMPT_PATH", - str(Path(__file__).resolve().parent.parent / "default_prompt.md"), -) +DEFAULT_PROMPT_PATH = os.environ.get("DEFAULT_PROMPT_PATH") # Tools stripped from the agent regardless of run state (none today: plan-mode # tool stripping is dynamic and handled by PlanModeMiddleware, not the profile). @@ -37,19 +35,28 @@ def _load_default_prompt() -> str: Returns empty string if the file doesn't exist or can't be read. """ try: - path = Path(DEFAULT_PROMPT_PATH) - if path.is_file(): - content = path.read_text().strip() - if content: - # Escape curly braces so .format() doesn't choke on them - escaped = content.replace("{", "{{").replace("}", "}}") - return f"""--- + if DEFAULT_PROMPT_PATH: + content = Path(DEFAULT_PROMPT_PATH).read_text().strip() + else: + content = ( + resources.files("agent.resources") + .joinpath("default_prompt.md") + .read_text(encoding="utf-8") + .strip() + ) + if content: + # Escape curly braces so .format() doesn't choke on them + escaped = content.replace("{", "{{").replace("}", "}}") + return f"""--- ### Custom Instructions {escaped}""" except Exception: - logger.warning("Failed to read default prompt file at %s", DEFAULT_PROMPT_PATH) + logger.warning( + "Failed to read default prompt from %s", + DEFAULT_PROMPT_PATH or "agent.resources/default_prompt.md", + ) return "" diff --git a/agent/resources/__init__.py b/agent/resources/__init__.py new file mode 100644 index 00000000..c9c2ef67 --- /dev/null +++ b/agent/resources/__init__.py @@ -0,0 +1 @@ +__all__: list[str] = [] diff --git a/default_prompt.md b/agent/resources/default_prompt.md similarity index 100% rename from default_prompt.md rename to agent/resources/default_prompt.md diff --git a/static/dark.svg b/assets/dark.svg similarity index 100% rename from static/dark.svg rename to assets/dark.svg diff --git a/static/light.svg b/assets/light.svg similarity index 100% rename from static/light.svg rename to assets/light.svg diff --git a/CUSTOMIZATION.md b/docs/CUSTOMIZATION.md similarity index 99% rename from CUSTOMIZATION.md rename to docs/CUSTOMIZATION.md index b6067eb4..b6165937 100644 --- a/CUSTOMIZATION.md +++ b/docs/CUSTOMIZATION.md @@ -439,7 +439,7 @@ Open SWE supports a `default_prompt.md` file for org-level instructions that app The file is loaded at agent startup and injected into the system prompt between the task overview and repository setup sections. -**Location:** [`default_prompt.md`](./default_prompt.md) in the project root. +**Location:** [`agent/resources/default_prompt.md`](../agent/resources/default_prompt.md) for the bundled default. **Override:** Set the `DEFAULT_PROMPT_PATH` environment variable to use a different file: diff --git a/INSTALLATION.md b/docs/INSTALLATION.md similarity index 100% rename from INSTALLATION.md rename to docs/INSTALLATION.md From 62d9945df468e032702778635ae048fdba8a1257 Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Fri, 17 Jul 2026 13:52:03 -0400 Subject: [PATCH 2/7] refactor: consolidate reviewer modules into agent/review/ MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Part of the domain-reorg adoption (build plan step C2): fork content, upstream layout. Nine 1:1 module moves (reviewer_diff/eval_store/ findings/groups/publish/reconcile/trace_context + review_style_ collector/guidance) into agent/review/, with internal relative imports re-wired to the new package depth. agent/review/__init__.py mirrors upstream's thin re-export shim (one of the 21 verified "A" structural adds). Rewrote the 38 grep hits across importer files (agent/{analyzer, ci_autofix,reviewer,webapp}.py, agent/dashboard/*, agent/middleware/ settle_review_check.py, agent/tools/*, agent/utils/github_feedback.py, agent/webhooks/github.py, evals/reviewer/*, and the reviewer test suite) to point at agent.review.*; 4 of the 38 hits were name collisions (list_reviewer_findings, reviewer_outcomes, _reviewer_thread_id, reviewer_thread_id — not the moved modules) and were left untouched. tests/test_github_checks.py's module-alias import (`from agent import reviewer_publish`) follows upstream's own `from agent.review import publish as reviewer_publish` pattern so downstream `reviewer_publish.*` call sites needed no changes. agent/reviewer.py and agent/webapp.py stay in place per the hard rule (fork content, import-only rewire) and are not part of this package. Gates: ruff check + ruff format --check, pytest --co -q (1637 collected), full unit suite (1637 passed), and the reviewer/findings suite in isolation (pytest -k "review or finding", 421 passed). --- agent/analyzer.py | 2 +- agent/ci_autofix.py | 2 +- agent/dashboard/agent_usage.py | 2 +- agent/dashboard/eval_jobs.py | 4 +- agent/dashboard/review_api.py | 6 +-- agent/dashboard/review_chat_api.py | 4 +- agent/dashboard/review_style_jobs.py | 2 +- agent/middleware/settle_review_check.py | 4 +- agent/review/__init__.py | 11 ++++ agent/{reviewer_diff.py => review/diff.py} | 4 +- .../eval_store.py} | 0 .../findings.py} | 0 .../{reviewer_groups.py => review/groups.py} | 4 +- .../publish.py} | 20 ++++---- .../reconcile.py} | 4 +- .../style_collector.py} | 0 .../style_guidance.py} | 0 .../trace_context.py} | 8 +-- agent/reviewer.py | 14 +++--- agent/tools/add_finding.py | 6 +-- agent/tools/list_findings.py | 4 +- agent/tools/list_review_findings.py | 2 +- agent/tools/publish_review.py | 10 ++-- agent/tools/reply_to_finding_thread.py | 4 +- agent/tools/resolve_finding_thread.py | 6 +-- agent/tools/update_finding.py | 2 +- agent/utils/github_feedback.py | 2 +- agent/webapp.py | 8 +-- agent/webhooks/github.py | 2 +- evals/reviewer/judge.py | 2 +- evals/reviewer/run_eval.py | 4 +- evals/reviewer/store_reporter.py | 2 +- evals/reviewer/target.py | 2 +- tests/test_dashboard_reviews.py | 2 +- tests/test_github_checks.py | 2 +- tests/test_review_style_collector.py | 2 +- tests/test_reviewer_diff.py | 6 +-- tests/test_reviewer_eval_target.py | 2 +- tests/test_reviewer_findings.py | 50 +++++++++---------- tests/test_reviewer_groups.py | 14 +++--- tests/test_reviewer_publish.py | 46 ++++++++--------- tests/test_reviewer_reconcile.py | 30 +++++------ tests/test_reviewer_tools.py | 4 +- tests/test_reviewer_trace_context.py | 10 ++-- 44 files changed, 163 insertions(+), 152 deletions(-) create mode 100644 agent/review/__init__.py rename agent/{reviewer_diff.py => review/diff.py} (98%) rename agent/{reviewer_eval_store.py => review/eval_store.py} (100%) rename agent/{reviewer_findings.py => review/findings.py} (100%) rename agent/{reviewer_groups.py => review/groups.py} (98%) rename agent/{reviewer_publish.py => review/publish.py} (99%) rename agent/{reviewer_reconcile.py => review/reconcile.py} (99%) rename agent/{review_style_collector.py => review/style_collector.py} (100%) rename agent/{review_style_guidance.py => review/style_guidance.py} (100%) rename agent/{reviewer_trace_context.py => review/trace_context.py} (98%) diff --git a/agent/analyzer.py b/agent/analyzer.py index 8f16934b..de5a6243 100644 --- a/agent/analyzer.py +++ b/agent/analyzer.py @@ -36,7 +36,7 @@ from .middleware import ( TimeoutWrapupMiddleware, ToolErrorMiddleware, ) -from .review_style_guidance import REVIEWER_STYLE_THEMES +from .review.style_guidance import REVIEWER_STYLE_THEMES from .server import ( DEFAULT_LLM_MAX_TOKENS, DEFAULT_LLM_MODEL_ID, diff --git a/agent/ci_autofix.py b/agent/ci_autofix.py index 13571f58..c9306656 100644 --- a/agent/ci_autofix.py +++ b/agent/ci_autofix.py @@ -26,7 +26,7 @@ from .dashboard.agent_overrides import load_profile, resolve_login_from_email_as from .dashboard.autofix_state import is_pr_autofix_disabled from .dashboard.enabled_repos import is_review_repo_enabled from .dispatch import dispatch_agent_run -from .reviewer_findings import REVIEWER_THREAD_KIND +from .review.findings import REVIEWER_THREAD_KIND from .utils.dashboard_links import dashboard_thread_url from .utils.github_app import get_github_app_installation_token from .utils.github_checks import post_autofix_status_check diff --git a/agent/dashboard/agent_usage.py b/agent/dashboard/agent_usage.py index f3a305c7..01840345 100644 --- a/agent/dashboard/agent_usage.py +++ b/agent/dashboard/agent_usage.py @@ -12,7 +12,7 @@ from typing import Any, Literal import httpx from langgraph_sdk import get_client -from ..reviewer_findings import REVIEWER_THREAD_KIND +from ..review.findings import REVIEWER_THREAD_KIND from ..utils.github_app import get_github_app_installation_token USAGE_THREAD_NAMESPACE: list[str] = ["agent_usage", "threads"] diff --git a/agent/dashboard/eval_jobs.py b/agent/dashboard/eval_jobs.py index d64c7ea1..91293ff4 100644 --- a/agent/dashboard/eval_jobs.py +++ b/agent/dashboard/eval_jobs.py @@ -17,13 +17,13 @@ from typing import Any, Literal, TypedDict from langgraph_sdk import get_client -from agent.reviewer_eval_store import ( +from agent.review.eval_store import ( _HEARTBEAT_STALE_SECONDS, DEFAULT_EVAL_PROJECT, EVALS_NAMESPACE, REVIEWER_EVAL_KEY, ) -from agent.reviewer_findings import REVIEW_FINDING_CAP +from agent.review.findings import REVIEW_FINDING_CAP logger = logging.getLogger(__name__) diff --git a/agent/dashboard/review_api.py b/agent/dashboard/review_api.py index 1c84aca6..c0db55c1 100644 --- a/agent/dashboard/review_api.py +++ b/agent/dashboard/review_api.py @@ -19,7 +19,7 @@ from urllib.parse import urljoin, urlparse import httpx from fastapi import HTTPException, Response -from ..reviewer_findings import REVIEWER_THREAD_KIND +from ..review.findings import REVIEWER_THREAD_KIND from ..utils.github_app import get_github_app_installation_token from ..utils.github_checks import github_headers from ..utils.thread_ops import langgraph_client @@ -443,7 +443,7 @@ async def create_review_comment( _HTML_COMMENT_RE = re.compile(r"", re.DOTALL) -# Inline comments the reviewer posts carry this hidden marker (see reviewer_publish). +# Inline comments the reviewer posts carry this hidden marker (see review.publish). _OPEN_SWE_COMMENT_RE = re.compile(r"