mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 11:33:14 +00:00
* fix: reset stale sandbox creation sentinel Co-authored-by: Johannes du Plessis <51395795+johannes117@users.noreply.github.com> * fix: treat SANDBOX_CREATING as a timestamped cross-process lock Only reset the sentinel when proven stale (older than the creation timeout); otherwise wait for the worker that holds the lock so a concurrent run does not create a duplicate sandbox. * feat(analyzer): outcomes dataset + bootstrap/continual split via skills Rename the review_style_analyzer graph to `analyzer` and split it into two modes, plus capture reviewer finding outcomes for continual learning. - Outcomes dataset: upsert resolved-by-commit (positive), dismissed (false positive), and GitHub/Slack thumbs findings into a single LangSmith dataset (openswe-reviewer-outcomes), keyed deterministically per finding+source. Emit points wired into update_finding, resolve_finding_thread, and the GitHub/Slack reaction handlers. - Two playbooks delivered as deepagents skills (bootstrap-repo-analysis, continual-learning), served as virtual files via a CompositeBackend /skills/ route + StateBackend (seeded into the run files channel at invoke time, never written to the sandbox). Mode is set by the launcher; continual runs fall back to the GitHub App installation token. - Split launcher into start_bootstrap_analysis + start_continual_run; register a per-repo nightly continual-learning cron when bootstrap completes. - New read_finding_outcomes tool feeds confirmed/dismissed findings back to the continual playbook. Tests for outcome label mapping, skills helper, and cron idempotency. * fix(analyzer): anchor continual cron runs to a real thread_id The nightly continual-learning cron is threadless, and get_analyzer early-returns an empty agent when configurable.thread_id is missing — so every cron-launched run no-op'd before reading outcomes or saving a refined prompt. Include the repo's deterministic analyzer thread_id in the continual run configurable so the run executes; the threadless run carries no message history, so nightly runs don't accumulate context. * refactor(analyzer): move cron lifecycle calls out of the review-styles store Drop the inline `analyzer_cron` imports from review_styles.py (added only to dodge a circular import) by relocating the cron-trigger calls to the layer above the store: registration to the save_review_style tool (after a prompt is saved) and removal to the dashboard delete route. review_styles.py is now a pure store again with top-level imports only. * refactor: hoist reviewer_outcomes imports to module level Move the two inline emit_finding_status_outcome imports introduced in this PR (update_finding, resolve_finding_thread) to top-level imports. reviewer_outcomes only depends on langsmith, so there is no circular import to avoid. --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
226 lines
6.6 KiB
Python
226 lines
6.6 KiB
Python
"""Slack reaction feedback handling."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import logging
|
|
import os
|
|
from typing import Any
|
|
|
|
from langgraph_sdk import get_client
|
|
from langgraph_sdk.client import LangGraphClient
|
|
|
|
from .langsmith import create_langsmith_feedback, delete_langsmith_feedback
|
|
from .reviewer_outcomes import outcome_from_score as _outcome_from_score
|
|
from .reviewer_outcomes import upsert_run_outcome
|
|
from .slack import lookup_slack_run_mapping
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
LANGGRAPH_URL = os.environ.get("LANGGRAPH_URL") or os.environ.get(
|
|
"LANGGRAPH_URL_PROD", "http://localhost:2024"
|
|
)
|
|
|
|
FEEDBACK_REACTIONS: dict[str, float] = {
|
|
"+1": 1.0,
|
|
"thumbsup": 1.0,
|
|
"-1": 0.0,
|
|
"thumbsdown": 0.0,
|
|
}
|
|
|
|
_REACTION_STATE_NAMESPACE = "slack_reaction_state"
|
|
_REACTION_EVENT_NAMESPACE = "slack_reaction_events"
|
|
|
|
|
|
def _read_active_reactions(item: dict[str, Any] | None) -> set[str]:
|
|
if not item:
|
|
return set()
|
|
value = item.get("value")
|
|
if not isinstance(value, dict):
|
|
return set()
|
|
reactions = value.get("reactions")
|
|
if not isinstance(reactions, list):
|
|
return set()
|
|
return {reaction for reaction in reactions if isinstance(reaction, str)}
|
|
|
|
|
|
def _feedback_key(channel_id: str, user_id: str, message_ts: str) -> str:
|
|
return f"slack_reaction:{channel_id}:{user_id}:{message_ts}"
|
|
|
|
|
|
def _reaction_state_key(run_id: str, user_id: str, message_ts: str) -> str:
|
|
return f"{run_id}:{user_id}:{message_ts}"
|
|
|
|
|
|
async def _event_was_processed(
|
|
langgraph_client: LangGraphClient, channel_id: str, event_id: str
|
|
) -> bool:
|
|
if not event_id:
|
|
return False
|
|
item = await langgraph_client.store.get_item((_REACTION_EVENT_NAMESPACE, channel_id), event_id)
|
|
return bool(item)
|
|
|
|
|
|
async def _mark_event_processed(
|
|
langgraph_client: LangGraphClient, channel_id: str, event_id: str
|
|
) -> None:
|
|
if not event_id:
|
|
return
|
|
await langgraph_client.store.put_item(
|
|
(_REACTION_EVENT_NAMESPACE, channel_id), event_id, {"event_id": event_id}
|
|
)
|
|
|
|
|
|
async def _update_reaction_state(
|
|
langgraph_client: LangGraphClient,
|
|
*,
|
|
channel_id: str,
|
|
run_id: str,
|
|
user_id: str,
|
|
message_ts: str,
|
|
reaction: str,
|
|
added: bool,
|
|
) -> set[str]:
|
|
namespace = (_REACTION_STATE_NAMESPACE, channel_id)
|
|
key = _reaction_state_key(run_id, user_id, message_ts)
|
|
item = await langgraph_client.store.get_item(namespace, key)
|
|
active_reactions = _read_active_reactions(item)
|
|
|
|
if added:
|
|
active_reactions.add(reaction)
|
|
else:
|
|
active_reactions.discard(reaction)
|
|
|
|
await langgraph_client.store.put_item(
|
|
namespace,
|
|
key,
|
|
{
|
|
"run_id": run_id,
|
|
"user_id": user_id,
|
|
"message_ts": message_ts,
|
|
"reactions": sorted(active_reactions),
|
|
},
|
|
)
|
|
return active_reactions
|
|
|
|
|
|
def _score_reactions(reactions: set[str]) -> float | None:
|
|
scores = {
|
|
FEEDBACK_REACTIONS[reaction] for reaction in reactions if reaction in FEEDBACK_REACTIONS
|
|
}
|
|
if not scores:
|
|
return None
|
|
if len(scores) > 1:
|
|
# Conflicting positive + negative reactions from the same user — treat
|
|
# as ambiguous and clear feedback rather than recording a misleading
|
|
# average score.
|
|
return None
|
|
return next(iter(scores))
|
|
|
|
|
|
async def process_slack_reaction(
|
|
event: dict[str, Any],
|
|
*,
|
|
event_id: str = "",
|
|
added: bool,
|
|
) -> None:
|
|
reaction = event.get("reaction")
|
|
if not isinstance(reaction, str) or reaction not in FEEDBACK_REACTIONS:
|
|
return
|
|
|
|
item = event.get("item")
|
|
if not isinstance(item, dict) or item.get("type") != "message":
|
|
return
|
|
|
|
channel_id = item.get("channel")
|
|
message_ts = item.get("ts")
|
|
user_id = event.get("user")
|
|
if not (
|
|
isinstance(channel_id, str)
|
|
and channel_id
|
|
and isinstance(message_ts, str)
|
|
and message_ts
|
|
and isinstance(user_id, str)
|
|
and user_id
|
|
):
|
|
return
|
|
|
|
langgraph_client = get_client(url=LANGGRAPH_URL)
|
|
if await _event_was_processed(langgraph_client, channel_id, event_id):
|
|
return
|
|
|
|
mapping = await lookup_slack_run_mapping(langgraph_client, channel_id, message_ts)
|
|
if not mapping:
|
|
logger.debug(
|
|
"No run mapping for Slack reaction on channel=%s message=%s",
|
|
channel_id,
|
|
message_ts,
|
|
)
|
|
return
|
|
run_id_value = mapping.get("run_id")
|
|
if not isinstance(run_id_value, str) or not run_id_value:
|
|
return
|
|
run_id = run_id_value
|
|
|
|
triggering_user_id = mapping.get("triggering_user_id")
|
|
if isinstance(triggering_user_id, str) and triggering_user_id and triggering_user_id != user_id:
|
|
# Only the user who triggered the run may give feedback on it. Other
|
|
# reactors are ignored to keep eval signal clean in shared channels.
|
|
logger.debug(
|
|
"Ignoring Slack reaction from non-triggering user=%s on run=%s",
|
|
user_id,
|
|
run_id,
|
|
)
|
|
return
|
|
|
|
active_reactions = await _update_reaction_state(
|
|
langgraph_client,
|
|
channel_id=channel_id,
|
|
run_id=run_id,
|
|
user_id=user_id,
|
|
message_ts=message_ts,
|
|
reaction=reaction,
|
|
added=added,
|
|
)
|
|
|
|
key = _feedback_key(channel_id, user_id, message_ts)
|
|
source_info = {
|
|
"source": "slack_reaction",
|
|
"channel_id": channel_id,
|
|
"message_ts": message_ts,
|
|
"user_id": user_id,
|
|
}
|
|
score = _score_reactions(active_reactions)
|
|
if score is None:
|
|
success = await asyncio.to_thread(delete_langsmith_feedback, run_id, key)
|
|
else:
|
|
success = await asyncio.to_thread(
|
|
create_langsmith_feedback,
|
|
run_id,
|
|
key,
|
|
score=score,
|
|
comment=f"Slack reaction feedback from user {user_id}",
|
|
source_info={**source_info, "reactions": sorted(active_reactions)},
|
|
)
|
|
|
|
outcome = _outcome_from_score(score, source="slack")
|
|
if outcome is not None:
|
|
label, label_source = outcome
|
|
await asyncio.to_thread(
|
|
upsert_run_outcome,
|
|
label=label,
|
|
label_source=label_source,
|
|
run_id=run_id,
|
|
extra={"channel_id": channel_id},
|
|
)
|
|
|
|
if success:
|
|
await _mark_event_processed(langgraph_client, channel_id, event_id)
|
|
|
|
|
|
async def process_slack_reaction_added(event: dict[str, Any], event_id: str = "") -> None:
|
|
await process_slack_reaction(event, event_id=event_id, added=True)
|
|
|
|
|
|
async def process_slack_reaction_removed(event: dict[str, Any], event_id: str = "") -> None:
|
|
await process_slack_reaction(event, event_id=event_id, added=False)
|