mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 20:53:15 +00:00
* Run reviewer eval in a GitHub Action; make dashboard a read-only progress view The dashboard launched the eval as a subprocess inside the serving deployment worker, so a container recycle killed long runs and discarded results that had already completed server-side. Move the harness to a workflow_dispatch Action (run on prod). run_eval now publishes status/progress/log-tail to the LangGraph store record the dashboard reads, so /admin/evals stays a live view; a killed Action surfaces as failed via the stale-heartbeat reconcile. * reviewer_eval workflow: pass inputs via env, no shell interpolation Addresses the reviewer finding: workflow_dispatch string inputs were interpolated into the run: block (limit unquoted), allowing shell injection in a job holding LANGSMITH/ANTHROPIC keys. Pass inputs through env and reference quoted "$VARS"; validate limit is numeric and build its flag in bash. --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
127 lines
4.8 KiB
YAML
127 lines
4.8 KiB
YAML
# Runs the reviewer benchmark (evals/reviewer/run_eval.py) on a durable runner instead of as a
|
|
# subprocess inside the serving deployment. Trigger it from the Actions UI / `gh workflow run`,
|
|
# selecting the `prod` branch so the harness + judge match the deployed reviewer it scores.
|
|
#
|
|
# Progress is streamed to the LangGraph store record the dashboard reads, so the run shows up live
|
|
# at /admin/evals. Required repository config:
|
|
# secrets: LANGSMITH_API_KEY, ANTHROPIC_API_KEY (judge runs in-process; reviewer model keys
|
|
# are NOT needed — the reviewer runs in the deployment)
|
|
# secret or var: LANGGRAPH_URL (the deployment URL the eval drives + reports to)
|
|
name: Reviewer eval
|
|
|
|
on:
|
|
workflow_dispatch:
|
|
inputs:
|
|
model_id:
|
|
description: Reviewer model id
|
|
type: string
|
|
default: google_genai:gemini-3.5-flash
|
|
reasoning_effort:
|
|
description: Reasoning effort
|
|
type: string
|
|
default: medium
|
|
dataset_name:
|
|
description: LangSmith dataset
|
|
type: string
|
|
default: openswe-reviewer-v1
|
|
experiment_prefix:
|
|
description: Run name (LangSmith experiment prefix)
|
|
type: string
|
|
default: openswe-review-confidence
|
|
max_concurrency:
|
|
description: Max concurrent PRs
|
|
type: string
|
|
default: "5"
|
|
score_mode:
|
|
description: all_findings | surfaced_findings
|
|
type: choice
|
|
default: all_findings
|
|
options:
|
|
- all_findings
|
|
- surfaced_findings
|
|
severity_threshold:
|
|
description: Severity threshold (surfaced_findings only)
|
|
type: choice
|
|
default: medium
|
|
options:
|
|
- low
|
|
- medium
|
|
- high
|
|
- critical
|
|
cap:
|
|
description: Max surfaced findings per PR (surfaced_findings only)
|
|
type: string
|
|
default: "4"
|
|
limit:
|
|
description: Run only the first N examples (blank = full dataset)
|
|
type: string
|
|
default: ""
|
|
langsmith_project:
|
|
description: LangSmith tracing project for eval traces
|
|
type: string
|
|
default: open-swe-evals
|
|
assistant_id:
|
|
description: Reviewer assistant id
|
|
type: string
|
|
default: reviewer
|
|
|
|
concurrency:
|
|
group: reviewer-eval
|
|
cancel-in-progress: false
|
|
|
|
permissions:
|
|
contents: read
|
|
|
|
jobs:
|
|
reviewer-eval:
|
|
name: Reviewer eval
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 360
|
|
steps:
|
|
- uses: actions/checkout@v6
|
|
- uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
|
|
- name: Install dependencies
|
|
run: uv sync --locked
|
|
- name: Run reviewer eval
|
|
# Inputs are passed via env and referenced as quoted "$VARS" — never
|
|
# interpolated into the script — so dispatcher-supplied text is treated
|
|
# as data, not shell syntax.
|
|
env:
|
|
LANGSMITH_API_KEY: ${{ secrets.LANGSMITH_API_KEY }}
|
|
LANGCHAIN_API_KEY: ${{ secrets.LANGSMITH_API_KEY }}
|
|
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
|
LANGGRAPH_URL: ${{ secrets.LANGGRAPH_URL || vars.LANGGRAPH_URL }}
|
|
REVIEWER_EVAL_REPORT_STORE: "1"
|
|
INPUT_MODEL_ID: ${{ inputs.model_id }}
|
|
INPUT_REASONING_EFFORT: ${{ inputs.reasoning_effort }}
|
|
INPUT_DATASET_NAME: ${{ inputs.dataset_name }}
|
|
INPUT_EXPERIMENT_PREFIX: ${{ inputs.experiment_prefix }}
|
|
INPUT_MAX_CONCURRENCY: ${{ inputs.max_concurrency }}
|
|
INPUT_SCORE_MODE: ${{ inputs.score_mode }}
|
|
INPUT_SEVERITY_THRESHOLD: ${{ inputs.severity_threshold }}
|
|
INPUT_CAP: ${{ inputs.cap }}
|
|
INPUT_LIMIT: ${{ inputs.limit }}
|
|
INPUT_LANGSMITH_PROJECT: ${{ inputs.langsmith_project }}
|
|
INPUT_ASSISTANT_ID: ${{ inputs.assistant_id }}
|
|
run: |
|
|
set -euo pipefail
|
|
limit_args=()
|
|
if [ -n "${INPUT_LIMIT}" ]; then
|
|
if ! [[ "${INPUT_LIMIT}" =~ ^[0-9]+$ ]]; then
|
|
echo "limit must be a positive integer, got: ${INPUT_LIMIT}" >&2
|
|
exit 1
|
|
fi
|
|
limit_args=(--limit "${INPUT_LIMIT}")
|
|
fi
|
|
uv run python -m evals.reviewer.run_eval \
|
|
--model-id "${INPUT_MODEL_ID}" \
|
|
--reasoning-effort "${INPUT_REASONING_EFFORT}" \
|
|
--dataset-name "${INPUT_DATASET_NAME}" \
|
|
--experiment-prefix "${INPUT_EXPERIMENT_PREFIX}" \
|
|
--max-concurrency "${INPUT_MAX_CONCURRENCY}" \
|
|
--score-mode "${INPUT_SCORE_MODE}" \
|
|
--severity-threshold "${INPUT_SEVERITY_THRESHOLD}" \
|
|
--cap "${INPUT_CAP}" \
|
|
--langsmith-project "${INPUT_LANGSMITH_PROJECT}" \
|
|
--assistant-id "${INPUT_ASSISTANT_ID}" \
|
|
"${limit_args[@]}"
|