mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-10-07 16:19:09 +00:00
Deployment-readiness for the Bedrock migration (PR #62): - instance-role.ts: least-privilege bedrock:InvokeModel[WithResponseStream] on the us.anthropic.claude-opus-4-8 inference-profile ARN + the foundation-model ARN in each routed region (us-east-1/2, us-west-2). The model runs in the server process on the box, so the EC2 instance role is the principal. Simulator-verified (allowed for opus-4-8, implicitDeny for other models) and synth-verified. Passed the mandatory GPT-4.1 IAM cross-review (no blockers, least-privilege confirmed). - config-store.ts: IaC SSM LLM_MODEL_ID anthropic:claude-opus-4-8 -> bedrock_converse:us.anthropic.claude-opus-4-8. This SSM value overrides seed_store.sh's default via pick precedence, so the seed-script fix alone was insufficient — both sources now point at the supported Bedrock id. - infra/README.md + evals/reviewer/config.toml: repoint stale anthropic:/google_genai: ids to the Bedrock id (config.toml's model_id was an active, now-broken value). AWS_REGION is already wired via user-data.sh (IMDS -> boot.env), so no change needed there.
23 lines
852 B
TOML
23 lines
852 B
TOML
dataset_name = "openswe-reviewer-v1"
|
|
experiment_prefix = "openswe-review-confidence"
|
|
max_concurrency = 5
|
|
|
|
# LangSmith tracing project for eval runs. Keeps eval traces out of the
|
|
# deployment's production project.
|
|
langsmith_project = "open-swe-evals"
|
|
|
|
# Leave blank to use LANGGRAPH_URL or local dev.
|
|
langgraph_url = ""
|
|
assistant_id = "reviewer"
|
|
# models (post Bedrock/Fireworks migration): bedrock_converse:us.anthropic.claude-opus-4-8,
|
|
# or any fireworks:* id in agent/dashboard/options.py SUPPORTED_MODELS.
|
|
model_id = "bedrock_converse:us.anthropic.claude-opus-4-8"
|
|
reasoning_effort = "medium"
|
|
|
|
# score_mode:
|
|
# - "all_findings" — score every add_finding the agent emits (no gating).
|
|
# - "surfaced_findings" — only findings that pass the production severity
|
|
# threshold and cap.
|
|
score_mode = "all_findings"
|
|
severity_threshold = "medium"
|
|
cap = 4
|