import { Navigate, createFileRoute } from "@tanstack/react-router"
import { useMutation, useQuery, useQueryClient } from "@tanstack/react-query"
import { useEffect, useRef, useState } from "react"
import type { ReactNode } from "react"
import type {
ModelOption,
ReviewerEvalConfig,
ReviewerEvalScoreMode,
ReviewerEvalSeverity,
ReviewerEvalStartRequest,
ReviewerEvalStatus,
} from "@/lib/api"
import { AppShell, SettingsSection } from "@/components/AppShell"
import { Button } from "@/components/ui/button"
import { Input } from "@/components/ui/input"
import {
Select,
SelectContent,
SelectItem,
SelectTrigger,
SelectValue,
} from "@/components/ui/select"
import { Skeleton } from "@/components/ui/skeleton"
import { api } from "@/lib/api"
import { useSession } from "@/lib/session"
import { cn } from "@/lib/utils"
export const Route = createFileRoute("/admin_/evals")({ component: ReviewerEvalPage })
function ReviewerEvalPage() {
const session = useSession()
if (session.isLoading) {
return (
)
}
if (!session.data) return
if (!session.data.is_admin) return
return (
)
}
const DEFAULT_REVIEWER_EVAL_CONFIG: ReviewerEvalConfig = {
dataset_name: "openswe-reviewer-v1",
experiment_prefix: "openswe-review-confidence",
max_concurrency: 5,
langsmith_project: "open-swe-evals",
langgraph_url: "",
assistant_id: "reviewer",
model_id: "google_genai:gemini-3.5-flash",
reasoning_effort: "medium",
score_mode: "all_findings",
severity_threshold: "medium",
cap: 4,
}
interface ReviewerEvalFormState {
dataset_name: string
experiment_prefix: string
max_concurrency: string
langsmith_project: string
langgraph_url: string
assistant_id: string
model_id: string
reasoning_effort: string
score_mode: ReviewerEvalScoreMode
severity_threshold: ReviewerEvalSeverity
cap: string
limit: string
}
function formFromConfig(
config: ReviewerEvalConfig,
limit: number | null = null
): ReviewerEvalFormState {
return {
dataset_name: config.dataset_name,
experiment_prefix: config.experiment_prefix,
max_concurrency: String(config.max_concurrency),
langsmith_project: config.langsmith_project,
langgraph_url: config.langgraph_url,
assistant_id: config.assistant_id,
model_id: config.model_id,
reasoning_effort: config.reasoning_effort,
score_mode: config.score_mode,
severity_threshold: config.severity_threshold,
cap: String(config.cap),
limit: limit ? String(limit) : "",
}
}
function parsePositiveInt(label: string, value: string): number {
const n = Number(value.trim())
if (!Number.isInteger(n) || n <= 0) {
throw new Error(`${label} must be a positive whole number`)
}
return n
}
function parseOptionalPositiveInt(label: string, value: string): number | null {
return value.trim() ? parsePositiveInt(label, value) : null
}
function parseNonNegativeInt(label: string, value: string): number {
const n = Number(value.trim())
if (!Number.isInteger(n) || n < 0) {
throw new Error(`${label} must be a non-negative whole number`)
}
return n
}
function requireText(label: string, value: string): string {
const text = value.trim()
if (!text) throw new Error(`${label} is required`)
return text
}
function FieldGroup({
label,
description,
children,
}: {
label: string
description?: string
children: ReactNode
}) {
return (
{label}
{description && (
{description}
)}
{children}
)
}
function Field({
label,
className,
children,
}: {
label: string
className?: string
children: ReactNode
}) {
return (
{label}
{children}
)
}
function ReviewerEvalRunner() {
const qc = useQueryClient()
const [draft, setDraft] = useState(() =>
formFromConfig(DEFAULT_REVIEWER_EVAL_CONFIG)
)
const [error, setError] = useState(null)
const initialized = useRef(false)
const status = useQuery({
queryKey: ["reviewerEval"],
queryFn: api.getReviewerEval,
refetchInterval: (query) =>
query.state.data?.status === "running" ? 5000 : false,
})
const options = useQuery({ queryKey: ["options"], queryFn: api.options })
const data = status.data
const running = data?.status === "running"
const currentModel: ModelOption | undefined =
options.data?.models.find((m) => m.id === draft.model_id) ??
options.data?.models[0]
useEffect(() => {
if (initialized.current || !data?.config_snapshot) return
initialized.current = true
setDraft(formFromConfig(data.config_snapshot, data.limit))
}, [data?.config_snapshot, data?.limit])
useEffect(() => {
if (!currentModel) return
if (currentModel.id !== draft.model_id) {
setDraft((current) => ({
...current,
model_id: currentModel.id,
reasoning_effort: currentModel.default_effort,
}))
return
}
if (!currentModel.efforts.includes(draft.reasoning_effort)) {
setDraft((current) => ({
...current,
reasoning_effort: currentModel.default_effort,
}))
}
}, [currentModel, draft.model_id, draft.reasoning_effort])
const setField = (
key: TKey,
value: ReviewerEvalFormState[TKey]
) => {
setDraft((current) => ({ ...current, [key]: value }))
}
const buildRequest = (): ReviewerEvalStartRequest => {
return {
dataset_name: requireText("Dataset", draft.dataset_name),
experiment_prefix: requireText("Run name", draft.experiment_prefix),
max_concurrency: parsePositiveInt("Max concurrency", draft.max_concurrency),
langsmith_project: requireText("LangSmith project", draft.langsmith_project),
langgraph_url: draft.langgraph_url.trim(),
assistant_id: requireText("Assistant ID", draft.assistant_id),
model_id: requireText("Model", draft.model_id),
reasoning_effort: requireText("Effort", draft.reasoning_effort),
score_mode: draft.score_mode,
severity_threshold: draft.severity_threshold,
cap: parseNonNegativeInt("Cap", draft.cap),
limit: parseOptionalPositiveInt("Limit", draft.limit),
}
}
const onSuccess = (next: ReviewerEvalStatus) => {
qc.setQueryData(["reviewerEval"], next)
setError(null)
}
const onError = (e: Error) => setError(e.message)
const start = useMutation({
mutationFn: () => {
return api.startReviewerEval(buildRequest())
},
onSuccess,
onError,
})
const cancel = useMutation({
mutationFn: () => api.cancelReviewerEval(),
onSuccess,
onError,
})
return (
<>
setField("experiment_prefix", e.target.value)}
/>
setField("dataset_name", e.target.value)}
/>
setField("limit", e.target.value)}
/>
setField("max_concurrency", e.target.value)}
/>
setField("cap", e.target.value)}
/>
setField("langsmith_project", e.target.value)}
/>
setField("langgraph_url", e.target.value)}
/>
setField("assistant_id", e.target.value)}
/>
Start
Only one reviewer eval can run at a time.
{running && (
)}
{error && {error}
}
>
)
}
function ReviewerEvalStatusView({ data }: { data: ReviewerEvalStatus | null }) {
if (!data) {
return (
Loading reviewer eval status…
)
}
const config = data.config_snapshot
return (
)
}
function StatusLine({
label,
value,
strong = false,
}: {
label: string
value: string | null | undefined
strong?: boolean
}) {
return (
{label}:{" "}
{value || "—"}
)
}
function ReviewerEvalLogs() {
const status = useQuery({
queryKey: ["reviewerEval"],
queryFn: api.getReviewerEval,
refetchInterval: (query) =>
query.state.data?.status === "running" ? 5000 : false,
})
const logTail = status.data?.log_tail ?? null
const running = status.data?.status === "running"
const scrollRef = useRef(null)
const [follow, setFollow] = useState(true)
const [copied, setCopied] = useState(false)
useEffect(() => {
if (follow && scrollRef.current) {
scrollRef.current.scrollTop = scrollRef.current.scrollHeight
}
}, [logTail, follow])
const copyLogs = async () => {
if (!logTail) return
await navigator.clipboard.writeText(logTail)
setCopied(true)
window.setTimeout(() => setCopied(false), 1500)
}
return (
}
>
{logTail ? (
{
const el = e.currentTarget
const atBottom =
el.scrollHeight - el.scrollTop - el.clientHeight < 24
setFollow(atBottom)
}}
className="max-h-[28rem] overflow-auto whitespace-pre-wrap break-words rounded-md bg-muted/50 p-3 font-mono text-xs text-foreground"
>
{logTail}
) : (
{running ? "Waiting for output…" : "No output yet. Run an eval to see logs here."}
)}
)
}