open-swe/agent/utils/github_feedback.py
Johannes du Plessis 4a55145bb1
feat: outcomes dataset + bootstrap/continual split via skills (#1365)
* fix: reset stale sandbox creation sentinel

Co-authored-by: Johannes du Plessis <51395795+johannes117@users.noreply.github.com>

* fix: treat SANDBOX_CREATING as a timestamped cross-process lock

Only reset the sentinel when proven stale (older than the creation
timeout); otherwise wait for the worker that holds the lock so a
concurrent run does not create a duplicate sandbox.

* feat(analyzer): outcomes dataset + bootstrap/continual split via skills

Rename the review_style_analyzer graph to `analyzer` and split it into two
modes, plus capture reviewer finding outcomes for continual learning.

- Outcomes dataset: upsert resolved-by-commit (positive), dismissed (false
  positive), and GitHub/Slack thumbs findings into a single LangSmith dataset
  (openswe-reviewer-outcomes), keyed deterministically per finding+source.
  Emit points wired into update_finding, resolve_finding_thread, and the
  GitHub/Slack reaction handlers.
- Two playbooks delivered as deepagents skills (bootstrap-repo-analysis,
  continual-learning), served as virtual files via a CompositeBackend /skills/
  route + StateBackend (seeded into the run files channel at invoke time, never
  written to the sandbox). Mode is set by the launcher; continual runs fall
  back to the GitHub App installation token.
- Split launcher into start_bootstrap_analysis + start_continual_run; register
  a per-repo nightly continual-learning cron when bootstrap completes.
- New read_finding_outcomes tool feeds confirmed/dismissed findings back to the
  continual playbook.

Tests for outcome label mapping, skills helper, and cron idempotency.

* fix(analyzer): anchor continual cron runs to a real thread_id

The nightly continual-learning cron is threadless, and get_analyzer
early-returns an empty agent when configurable.thread_id is missing — so
every cron-launched run no-op'd before reading outcomes or saving a refined
prompt. Include the repo's deterministic analyzer thread_id in the continual
run configurable so the run executes; the threadless run carries no message
history, so nightly runs don't accumulate context.

* refactor(analyzer): move cron lifecycle calls out of the review-styles store

Drop the inline `analyzer_cron` imports from review_styles.py (added only to
dodge a circular import) by relocating the cron-trigger calls to the layer
above the store: registration to the save_review_style tool (after a prompt is
saved) and removal to the dashboard delete route. review_styles.py is now a
pure store again with top-level imports only.

* refactor: hoist reviewer_outcomes imports to module level

Move the two inline emit_finding_status_outcome imports introduced in this PR
(update_finding, resolve_finding_thread) to top-level imports. reviewer_outcomes
only depends on langsmith, so there is no circular import to avoid.

---------

Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
2026-06-01 13:25:12 -07:00

250 lines
7.7 KiB
Python

from __future__ import annotations
import asyncio
import logging
import os
import re
import uuid
from typing import Any
from langgraph_sdk import get_client
from langgraph_sdk.client import LangGraphClient
from ..reviewer_findings import list_findings
from .langsmith import create_langsmith_feedback, delete_langsmith_feedback
from .reviewer_outcomes import outcome_from_score, upsert_finding_outcome
logger = logging.getLogger(__name__)
LANGGRAPH_URL = os.environ.get("LANGGRAPH_URL") or os.environ.get(
"LANGGRAPH_URL_PROD", "http://localhost:2024"
)
GITHUB_FEEDBACK_REACTIONS: dict[str, float] = {
"+1": 1.0,
"-1": 0.0,
}
_REACTION_STATE_NAMESPACE = "github_reaction_state"
_REACTION_EVENT_NAMESPACE = "github_reaction_events"
_PULL_URL_RE = re.compile(r"/pulls/(\d+)\Z")
def _reviewer_thread_id(owner: str, repo: str, pr_number: int) -> str:
return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{owner}/{repo}/pr/{pr_number}/reviewer"))
def _read_active_reactions(item: dict[str, Any] | None) -> set[str]:
if not item:
return set()
value = item.get("value")
if not isinstance(value, dict):
return set()
reactions = value.get("reactions")
if not isinstance(reactions, list):
return set()
return {reaction for reaction in reactions if isinstance(reaction, str)}
def _reaction_state_key(run_id: str, user_login: str, comment_id: int) -> str:
return f"{run_id}:{user_login}:{comment_id}"
def _feedback_key(owner: str, repo: str, user_login: str, comment_id: int) -> str:
return f"github_reaction:{owner}/{repo}:{user_login}:{comment_id}"
def _score_reactions(reactions: set[str]) -> float | None:
scores = {
GITHUB_FEEDBACK_REACTIONS[reaction]
for reaction in reactions
if reaction in GITHUB_FEEDBACK_REACTIONS
}
if len(scores) != 1:
return None
return next(iter(scores))
def _extract_pr_number(payload: dict[str, Any]) -> int | None:
pull_request = payload.get("pull_request")
if isinstance(pull_request, dict) and isinstance(pull_request.get("number"), int):
return pull_request["number"]
comment = payload.get("comment")
if isinstance(comment, dict):
url = comment.get("pull_request_url")
if isinstance(url, str):
match = _PULL_URL_RE.search(url)
if match:
return int(match.group(1))
return None
async def _event_was_processed(
langgraph_client: LangGraphClient, repo_key: str, event_id: str
) -> bool:
if not event_id:
return False
item = await langgraph_client.store.get_item((_REACTION_EVENT_NAMESPACE, repo_key), event_id)
return bool(item)
async def _mark_event_processed(
langgraph_client: LangGraphClient, repo_key: str, event_id: str
) -> None:
if not event_id:
return
await langgraph_client.store.put_item(
(_REACTION_EVENT_NAMESPACE, repo_key), event_id, {"event_id": event_id}
)
async def _update_reaction_state(
langgraph_client: LangGraphClient,
*,
repo_key: str,
run_id: str,
user_login: str,
comment_id: int,
reaction: str,
added: bool,
) -> set[str]:
namespace = (_REACTION_STATE_NAMESPACE, repo_key)
key = _reaction_state_key(run_id, user_login, comment_id)
item = await langgraph_client.store.get_item(namespace, key)
active_reactions = _read_active_reactions(item)
if added:
active_reactions.add(reaction)
else:
active_reactions.discard(reaction)
if not active_reactions:
await langgraph_client.store.delete_item(namespace, key)
return active_reactions
await langgraph_client.store.put_item(
namespace,
key,
{
"run_id": run_id,
"user_login": user_login,
"comment_id": comment_id,
"reactions": sorted(active_reactions),
},
)
return active_reactions
async def process_github_reaction(
payload: dict[str, Any],
*,
delivery_id: str = "",
added: bool,
) -> None:
reaction = payload.get("reaction")
content = reaction.get("content") if isinstance(reaction, dict) else None
if not isinstance(content, str) or content not in GITHUB_FEEDBACK_REACTIONS:
return
comment = payload.get("comment")
comment_id = comment.get("id") if isinstance(comment, dict) else None
if not isinstance(comment_id, int):
return
repo = payload.get("repository")
owner = repo.get("owner", {}).get("login") if isinstance(repo, dict) else None
repo_name = repo.get("name") if isinstance(repo, dict) else None
pr_number = _extract_pr_number(payload)
sender = payload.get("sender")
user_login = sender.get("login") if isinstance(sender, dict) else None
if not (
isinstance(owner, str)
and owner
and isinstance(repo_name, str)
and repo_name
and isinstance(pr_number, int)
and isinstance(user_login, str)
and user_login
):
return
langgraph_client = get_client(url=LANGGRAPH_URL)
repo_key = f"{owner}/{repo_name}"
if await _event_was_processed(langgraph_client, repo_key, delivery_id):
return
thread_id = _reviewer_thread_id(owner, repo_name, pr_number)
findings = await list_findings(thread_id)
finding = next(
(
candidate
for candidate in findings
if candidate.get("github_review_comment_id") == comment_id
),
None,
)
if finding is None:
logger.debug("No tracked finding for GitHub review comment id %s", comment_id)
return
run_id = finding.get("github_review_run_id")
if not isinstance(run_id, str) or not run_id:
logger.debug("Finding %s has no LangSmith run id for feedback", finding.get("id"))
return
active_reactions = await _update_reaction_state(
langgraph_client,
repo_key=repo_key,
run_id=run_id,
user_login=user_login,
comment_id=comment_id,
reaction=content,
added=added,
)
key = _feedback_key(owner, repo_name, user_login, comment_id)
source_info = {
"source": "github_review_reaction",
"owner": owner,
"repo": repo_name,
"pr_number": pr_number,
"comment_id": comment_id,
"finding_id": finding.get("id"),
"user_login": user_login,
}
score = _score_reactions(active_reactions)
if score is None:
success = await asyncio.to_thread(delete_langsmith_feedback, run_id, key)
else:
success = await asyncio.to_thread(
create_langsmith_feedback,
run_id,
key,
score=score,
comment=f"GitHub review reaction feedback from {user_login}",
source_info={**source_info, "reactions": sorted(active_reactions)},
)
outcome = outcome_from_score(score, source="github")
if outcome is not None:
label, label_source = outcome
await asyncio.to_thread(
upsert_finding_outcome,
finding,
label=label,
label_source=label_source,
repo=repo_key,
pr_number=pr_number,
pr_url=f"https://github.com/{repo_key}/pull/{pr_number}",
head_sha=str(finding.get("first_seen_sha") or ""),
run_id=run_id,
thread_id=thread_id,
)
if success:
await _mark_event_processed(langgraph_client, repo_key, delivery_id)
async def process_github_reaction_added(payload: dict[str, Any], delivery_id: str = "") -> None:
await process_github_reaction(payload, delivery_id=delivery_id, added=True)
async def process_github_reaction_removed(payload: dict[str, Any], delivery_id: str = "") -> None:
await process_github_reaction(payload, delivery_id=delivery_id, added=False)