mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 05:43:14 +00:00
* fix: reset stale sandbox creation sentinel Co-authored-by: Johannes du Plessis <51395795+johannes117@users.noreply.github.com> * fix: treat SANDBOX_CREATING as a timestamped cross-process lock Only reset the sentinel when proven stale (older than the creation timeout); otherwise wait for the worker that holds the lock so a concurrent run does not create a duplicate sandbox. * feat(analyzer): outcomes dataset + bootstrap/continual split via skills Rename the review_style_analyzer graph to `analyzer` and split it into two modes, plus capture reviewer finding outcomes for continual learning. - Outcomes dataset: upsert resolved-by-commit (positive), dismissed (false positive), and GitHub/Slack thumbs findings into a single LangSmith dataset (openswe-reviewer-outcomes), keyed deterministically per finding+source. Emit points wired into update_finding, resolve_finding_thread, and the GitHub/Slack reaction handlers. - Two playbooks delivered as deepagents skills (bootstrap-repo-analysis, continual-learning), served as virtual files via a CompositeBackend /skills/ route + StateBackend (seeded into the run files channel at invoke time, never written to the sandbox). Mode is set by the launcher; continual runs fall back to the GitHub App installation token. - Split launcher into start_bootstrap_analysis + start_continual_run; register a per-repo nightly continual-learning cron when bootstrap completes. - New read_finding_outcomes tool feeds confirmed/dismissed findings back to the continual playbook. Tests for outcome label mapping, skills helper, and cron idempotency. * fix(analyzer): anchor continual cron runs to a real thread_id The nightly continual-learning cron is threadless, and get_analyzer early-returns an empty agent when configurable.thread_id is missing — so every cron-launched run no-op'd before reading outcomes or saving a refined prompt. Include the repo's deterministic analyzer thread_id in the continual run configurable so the run executes; the threadless run carries no message history, so nightly runs don't accumulate context. * refactor(analyzer): move cron lifecycle calls out of the review-styles store Drop the inline `analyzer_cron` imports from review_styles.py (added only to dodge a circular import) by relocating the cron-trigger calls to the layer above the store: registration to the save_review_style tool (after a prompt is saved) and removal to the dashboard delete route. review_styles.py is now a pure store again with top-level imports only. * refactor: hoist reviewer_outcomes imports to module level Move the two inline emit_finding_status_outcome imports introduced in this PR (update_finding, resolve_finding_thread) to top-level imports. reviewer_outcomes only depends on langsmith, so there is no circular import to avoid. --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
150 lines
5.7 KiB
Python
150 lines
5.7 KiB
Python
"""Analyzer graph.
|
|
|
|
Learns a per-repo review-style prompt for the reviewer agent. It mines
|
|
historical human PR review feedback and this reviewer's own past finding
|
|
outcomes (resolved / dismissed / 👍👎) to teach what this team flags and skips.
|
|
|
|
Uses the same sandbox + ``gh`` pattern as the reviewer agent. The dashboard
|
|
user's OAuth token is injected into the LangSmith GitHub proxy so ``gh`` works
|
|
on public repos even when the GitHub App is not installed on them.
|
|
"""
|
|
# ruff: noqa: E402
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import logging
|
|
import os
|
|
import warnings
|
|
|
|
from langgraph.graph.state import RunnableConfig
|
|
from langgraph.pregel import Pregel
|
|
|
|
warnings.filterwarnings("ignore", module="langchain_core._api.deprecation")
|
|
warnings.filterwarnings("ignore", message=".*Pydantic V1.*", category=UserWarning)
|
|
|
|
from deepagents import create_deep_agent
|
|
from deepagents.backends.composite import CompositeBackend
|
|
from deepagents.backends.protocol import SandboxBackendProtocol
|
|
from deepagents.backends.state import StateBackend
|
|
from langchain.agents.middleware import ModelCallLimitMiddleware
|
|
|
|
from .integrations.langsmith import _configure_github_proxy
|
|
from .middleware import SanitizeToolInputsMiddleware, ToolErrorMiddleware
|
|
from .review_style_guidance import REVIEWER_STYLE_THEMES
|
|
from .server import (
|
|
DEFAULT_LLM_MAX_TOKENS,
|
|
DEFAULT_LLM_MODEL_ID,
|
|
DEFAULT_RECURSION_LIMIT,
|
|
ensure_sandbox_for_thread,
|
|
graph_loaded_for_execution,
|
|
)
|
|
from .tools.read_finding_outcomes import read_finding_outcomes
|
|
from .tools.save_review_style import save_review_style_prompt
|
|
from .utils.analyzer_skills import SKILLS_ROUTE, skill_path_for_mode
|
|
from .utils.github_app import get_github_app_installation_token
|
|
from .utils.model import DEFAULT_LLM_REASONING, make_model, provider_model_kwargs
|
|
from .utils.sandbox_paths import aresolve_sandbox_work_dir
|
|
from .utils.sandbox_state import unwrap_sandbox_backend
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
STYLE_ANALYZER_MODEL_CALL_LIMIT = 80
|
|
|
|
# The per-mode procedure lives in the bundled SKILL.md playbooks (agent/skills/).
|
|
# This base prompt only orients the agent and points it at the right skill.
|
|
STYLE_ANALYZER_PROMPT = """You are a code-review style analyst for `{repo_owner}/{repo_name}`.
|
|
|
|
Sandbox: `{working_dir}`. Use the shell (``execute``) to run GitHub commands.
|
|
**Always invoke gh as:** `GH_TOKEN=dummy gh <command>`.
|
|
|
|
Your job is to produce/refine the per-repo review-style prompt and persist it with
|
|
`save_review_style_prompt`.
|
|
|
|
# Run mode: {mode}
|
|
|
|
Read and follow the playbook for this mode, then proceed:
|
|
|
|
read_file("{skill_path}", limit=1000)
|
|
|
|
Do not improvise the procedure — the skill is authoritative for how to gather
|
|
evidence and what to save.
|
|
|
|
# Alignment with our reviewer agent
|
|
|
|
{reviewer_themes}
|
|
"""
|
|
|
|
|
|
async def _configure_sandbox_github_proxy(
|
|
sandbox_backend: SandboxBackendProtocol,
|
|
github_token: str,
|
|
) -> None:
|
|
if os.getenv("SANDBOX_TYPE", "langsmith") != "langsmith":
|
|
return
|
|
backend = unwrap_sandbox_backend(sandbox_backend)
|
|
await asyncio.to_thread(_configure_github_proxy, backend.id, github_token)
|
|
|
|
|
|
async def get_analyzer(config: RunnableConfig) -> Pregel:
|
|
thread_id = config["configurable"].get("thread_id")
|
|
config["recursion_limit"] = DEFAULT_RECURSION_LIMIT
|
|
|
|
if thread_id is None or not graph_loaded_for_execution(config):
|
|
return create_deep_agent(system_prompt="", tools=[]).with_config(config)
|
|
|
|
sandbox_backend = await ensure_sandbox_for_thread(thread_id)
|
|
work_dir = await aresolve_sandbox_work_dir(sandbox_backend)
|
|
|
|
configurable = config["configurable"]
|
|
full_name = str(configurable.get("review_style_full_name") or "owner/repo")
|
|
owner, _, name = full_name.partition("/")
|
|
samples_text = str(configurable.get("review_style_samples_text") or "")
|
|
mode = str(configurable.get("analyzer_mode") or "bootstrap")
|
|
|
|
github_token = configurable.get("review_style_github_token")
|
|
if not (isinstance(github_token, str) and github_token):
|
|
# Nightly continual runs have no fresh dashboard OAuth token; fall back to
|
|
# the GitHub App installation token so `gh` still works through the proxy.
|
|
github_token = await get_github_app_installation_token()
|
|
if isinstance(github_token, str) and github_token:
|
|
await _configure_sandbox_github_proxy(sandbox_backend, github_token)
|
|
|
|
# Skills are served from a virtual StateBackend route; gh/clone/execute stay on
|
|
# the sandbox. SKILL.md files are seeded into the `files` channel at invoke time.
|
|
backend = CompositeBackend(default=sandbox_backend, routes={SKILLS_ROUTE: StateBackend()})
|
|
|
|
model_id = DEFAULT_LLM_MODEL_ID
|
|
model_kwargs = provider_model_kwargs(
|
|
model_id,
|
|
None,
|
|
max_tokens=DEFAULT_LLM_MAX_TOKENS,
|
|
openai_reasoning_default=DEFAULT_LLM_REASONING,
|
|
)
|
|
|
|
system_prompt = STYLE_ANALYZER_PROMPT.format(
|
|
repo_owner=owner or "<owner>",
|
|
repo_name=name or "<repo>",
|
|
working_dir=work_dir,
|
|
mode=mode,
|
|
skill_path=skill_path_for_mode(mode),
|
|
reviewer_themes=REVIEWER_STYLE_THEMES.strip(),
|
|
)
|
|
user_context = f"Repository: `{full_name}`\n\n{samples_text}".strip()
|
|
system_prompt = f"{system_prompt}\n\n{user_context}"
|
|
|
|
return create_deep_agent(
|
|
model=make_model(model_id, **model_kwargs),
|
|
system_prompt=system_prompt,
|
|
tools=[save_review_style_prompt, read_finding_outcomes],
|
|
backend=backend,
|
|
skills=[SKILLS_ROUTE],
|
|
middleware=[
|
|
SanitizeToolInputsMiddleware(),
|
|
ModelCallLimitMiddleware(
|
|
run_limit=STYLE_ANALYZER_MODEL_CALL_LIMIT,
|
|
exit_behavior="end",
|
|
),
|
|
ToolErrorMiddleware(),
|
|
],
|
|
).with_config(config)
|