mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 20:53:15 +00:00
Bumps [astral-sh/setup-uv](https://github.com/astral-sh/setup-uv) from 8.3.2 to 9.0.0.
- [Release notes](https://github.com/astral-sh/setup-uv/releases)
- [Commits](11f9893b08...c771a70e62)
---
updated-dependencies:
- dependency-name: astral-sh/setup-uv
dependency-version: 9.0.0
dependency-type: direct:production
update-type: version-update:semver-major
...
Signed-off-by: dependabot[bot] <support@github.com>
Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>
127 lines
4.8 KiB
YAML
127 lines
4.8 KiB
YAML
# Runs the reviewer benchmark (evals/reviewer/run_eval.py) on a durable runner instead of as a
|
|
# subprocess inside the serving deployment. Trigger it from the Actions UI / `gh workflow run`,
|
|
# selecting the `prod` branch so the harness + judge match the deployed reviewer it scores.
|
|
#
|
|
# Progress is streamed to the LangGraph store record the dashboard reads, so the run shows up live
|
|
# at /admin/evals. Required repository config:
|
|
# secrets: LANGSMITH_API_KEY, ANTHROPIC_API_KEY (judge runs in-process; reviewer model keys
|
|
# are NOT needed — the reviewer runs in the deployment)
|
|
# secret or var: LANGGRAPH_URL (the deployment URL the eval drives + reports to)
|
|
name: Reviewer eval
|
|
|
|
on:
|
|
workflow_dispatch:
|
|
inputs:
|
|
model_id:
|
|
description: Reviewer model id
|
|
type: string
|
|
default: google_genai:gemini-3.5-flash
|
|
reasoning_effort:
|
|
description: Reasoning effort
|
|
type: string
|
|
default: medium
|
|
dataset_name:
|
|
description: LangSmith dataset
|
|
type: string
|
|
default: openswe-reviewer-v1
|
|
experiment_prefix:
|
|
description: Run name (LangSmith experiment prefix)
|
|
type: string
|
|
default: openswe-review-confidence
|
|
max_concurrency:
|
|
description: Max concurrent PRs
|
|
type: string
|
|
default: "5"
|
|
score_mode:
|
|
description: all_findings | surfaced_findings
|
|
type: choice
|
|
default: surfaced_findings
|
|
options:
|
|
- all_findings
|
|
- surfaced_findings
|
|
severity_threshold:
|
|
description: Severity threshold (surfaced_findings only)
|
|
type: choice
|
|
default: low
|
|
options:
|
|
- low
|
|
- medium
|
|
- high
|
|
- critical
|
|
cap:
|
|
description: Max surfaced findings per PR (surfaced_findings only)
|
|
type: string
|
|
default: "6"
|
|
limit:
|
|
description: Run only the first N examples (blank = full dataset)
|
|
type: string
|
|
default: ""
|
|
langsmith_project:
|
|
description: LangSmith tracing project for eval traces
|
|
type: string
|
|
default: open-swe-evals
|
|
assistant_id:
|
|
description: Reviewer assistant id
|
|
type: string
|
|
default: reviewer
|
|
|
|
concurrency:
|
|
group: reviewer-eval
|
|
cancel-in-progress: false
|
|
|
|
permissions:
|
|
contents: read
|
|
|
|
jobs:
|
|
reviewer-eval:
|
|
name: Reviewer eval
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 360
|
|
steps:
|
|
- uses: actions/checkout@v7
|
|
- uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
|
- name: Install dependencies
|
|
run: uv sync --locked
|
|
- name: Run reviewer eval
|
|
# Inputs are passed via env and referenced as quoted "$VARS" — never
|
|
# interpolated into the script — so dispatcher-supplied text is treated
|
|
# as data, not shell syntax.
|
|
env:
|
|
LANGSMITH_API_KEY: ${{ secrets.LANGSMITH_API_KEY }}
|
|
LANGCHAIN_API_KEY: ${{ secrets.LANGSMITH_API_KEY }}
|
|
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
|
LANGGRAPH_URL: ${{ secrets.LANGGRAPH_URL || vars.LANGGRAPH_URL }}
|
|
REVIEWER_EVAL_REPORT_STORE: "1"
|
|
INPUT_MODEL_ID: ${{ inputs.model_id }}
|
|
INPUT_REASONING_EFFORT: ${{ inputs.reasoning_effort }}
|
|
INPUT_DATASET_NAME: ${{ inputs.dataset_name }}
|
|
INPUT_EXPERIMENT_PREFIX: ${{ inputs.experiment_prefix }}
|
|
INPUT_MAX_CONCURRENCY: ${{ inputs.max_concurrency }}
|
|
INPUT_SCORE_MODE: ${{ inputs.score_mode }}
|
|
INPUT_SEVERITY_THRESHOLD: ${{ inputs.severity_threshold }}
|
|
INPUT_CAP: ${{ inputs.cap }}
|
|
INPUT_LIMIT: ${{ inputs.limit }}
|
|
INPUT_LANGSMITH_PROJECT: ${{ inputs.langsmith_project }}
|
|
INPUT_ASSISTANT_ID: ${{ inputs.assistant_id }}
|
|
run: |
|
|
set -euo pipefail
|
|
limit_args=()
|
|
if [ -n "${INPUT_LIMIT}" ]; then
|
|
if ! [[ "${INPUT_LIMIT}" =~ ^[0-9]+$ ]]; then
|
|
echo "limit must be a positive integer, got: ${INPUT_LIMIT}" >&2
|
|
exit 1
|
|
fi
|
|
limit_args=(--limit "${INPUT_LIMIT}")
|
|
fi
|
|
uv run python -m evals.reviewer.run_eval \
|
|
--model-id "${INPUT_MODEL_ID}" \
|
|
--reasoning-effort "${INPUT_REASONING_EFFORT}" \
|
|
--dataset-name "${INPUT_DATASET_NAME}" \
|
|
--experiment-prefix "${INPUT_EXPERIMENT_PREFIX}" \
|
|
--max-concurrency "${INPUT_MAX_CONCURRENCY}" \
|
|
--score-mode "${INPUT_SCORE_MODE}" \
|
|
--severity-threshold "${INPUT_SEVERITY_THRESHOLD}" \
|
|
--cap "${INPUT_CAP}" \
|
|
--langsmith-project "${INPUT_LANGSMITH_PROJECT}" \
|
|
--assistant-id "${INPUT_ASSISTANT_ID}" \
|
|
"${limit_args[@]}"
|