open-swe/.github/workflows/reviewer-eval.yml
dependabot[bot] 678ede8008
Some checks are pending
CI / Lint (push) Waiting to run
CI / Format check (push) Waiting to run
CI / Typecheck (push) Waiting to run
CI / Unit tests (push) Waiting to run
CI / Playwright E2E (push) Waiting to run
CI / Docker build smoke (push) Waiting to run
CI / Triage ledger up to date (push) Waiting to run
CI / ui bun.lock in sync (push) Waiting to run
chore(deps): bump astral-sh/setup-uv from 9.0.0 to 10.0.1 (#261)
Bumps [astral-sh/setup-uv](https://github.com/astral-sh/setup-uv) from 9.0.0 to 10.0.1.
- [Release notes](https://github.com/astral-sh/setup-uv/releases)
- [Commits](c771a70e62...20cfd1bf94)

---
updated-dependencies:
- dependency-name: astral-sh/setup-uv
  dependency-version: 10.0.1
  dependency-type: direct:production
  update-type: version-update:semver-major
...

Signed-off-by: dependabot[bot] <support@github.com>
Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>
2026-08-20 17:04:36 +00:00

127 lines
4.8 KiB
YAML

# Runs the reviewer benchmark (evals/reviewer/run_eval.py) on a durable runner instead of as a
# subprocess inside the serving deployment. Trigger it from the Actions UI / `gh workflow run`,
# selecting the `prod` branch so the harness + judge match the deployed reviewer it scores.
#
# Progress is streamed to the LangGraph store record the dashboard reads, so the run shows up live
# at /admin/evals. Required repository config:
# secrets: LANGSMITH_API_KEY, ANTHROPIC_API_KEY (judge runs in-process; reviewer model keys
# are NOT needed — the reviewer runs in the deployment)
# secret or var: LANGGRAPH_URL (the deployment URL the eval drives + reports to)
name: Reviewer eval
on:
workflow_dispatch:
inputs:
model_id:
description: Reviewer model id
type: string
default: google_genai:gemini-3.5-flash
reasoning_effort:
description: Reasoning effort
type: string
default: medium
dataset_name:
description: LangSmith dataset
type: string
default: openswe-reviewer-v1
experiment_prefix:
description: Run name (LangSmith experiment prefix)
type: string
default: openswe-review-confidence
max_concurrency:
description: Max concurrent PRs
type: string
default: "5"
score_mode:
description: all_findings | surfaced_findings
type: choice
default: surfaced_findings
options:
- all_findings
- surfaced_findings
severity_threshold:
description: Severity threshold (surfaced_findings only)
type: choice
default: low
options:
- low
- medium
- high
- critical
cap:
description: Max surfaced findings per PR (surfaced_findings only)
type: string
default: "6"
limit:
description: Run only the first N examples (blank = full dataset)
type: string
default: ""
langsmith_project:
description: LangSmith tracing project for eval traces
type: string
default: open-swe-evals
assistant_id:
description: Reviewer assistant id
type: string
default: reviewer
concurrency:
group: reviewer-eval
cancel-in-progress: false
permissions:
contents: read
jobs:
reviewer-eval:
name: Reviewer eval
runs-on: ubuntu-latest
timeout-minutes: 360
steps:
- uses: actions/checkout@v7
- uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
- name: Install dependencies
run: uv sync --locked
- name: Run reviewer eval
# Inputs are passed via env and referenced as quoted "$VARS" — never
# interpolated into the script — so dispatcher-supplied text is treated
# as data, not shell syntax.
env:
LANGSMITH_API_KEY: ${{ secrets.LANGSMITH_API_KEY }}
LANGCHAIN_API_KEY: ${{ secrets.LANGSMITH_API_KEY }}
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
LANGGRAPH_URL: ${{ secrets.LANGGRAPH_URL || vars.LANGGRAPH_URL }}
REVIEWER_EVAL_REPORT_STORE: "1"
INPUT_MODEL_ID: ${{ inputs.model_id }}
INPUT_REASONING_EFFORT: ${{ inputs.reasoning_effort }}
INPUT_DATASET_NAME: ${{ inputs.dataset_name }}
INPUT_EXPERIMENT_PREFIX: ${{ inputs.experiment_prefix }}
INPUT_MAX_CONCURRENCY: ${{ inputs.max_concurrency }}
INPUT_SCORE_MODE: ${{ inputs.score_mode }}
INPUT_SEVERITY_THRESHOLD: ${{ inputs.severity_threshold }}
INPUT_CAP: ${{ inputs.cap }}
INPUT_LIMIT: ${{ inputs.limit }}
INPUT_LANGSMITH_PROJECT: ${{ inputs.langsmith_project }}
INPUT_ASSISTANT_ID: ${{ inputs.assistant_id }}
run: |
set -euo pipefail
limit_args=()
if [ -n "${INPUT_LIMIT}" ]; then
if ! [[ "${INPUT_LIMIT}" =~ ^[0-9]+$ ]]; then
echo "limit must be a positive integer, got: ${INPUT_LIMIT}" >&2
exit 1
fi
limit_args=(--limit "${INPUT_LIMIT}")
fi
uv run python -m evals.reviewer.run_eval \
--model-id "${INPUT_MODEL_ID}" \
--reasoning-effort "${INPUT_REASONING_EFFORT}" \
--dataset-name "${INPUT_DATASET_NAME}" \
--experiment-prefix "${INPUT_EXPERIMENT_PREFIX}" \
--max-concurrency "${INPUT_MAX_CONCURRENCY}" \
--score-mode "${INPUT_SCORE_MODE}" \
--severity-threshold "${INPUT_SEVERITY_THRESHOLD}" \
--cap "${INPUT_CAP}" \
--langsmith-project "${INPUT_LANGSMITH_PROJECT}" \
--assistant-id "${INPUT_ASSISTANT_ID}" \
"${limit_args[@]}"