import { Navigate, createFileRoute } from "@tanstack/react-router" import { useMutation, useQuery, useQueryClient } from "@tanstack/react-query" import { useEffect, useRef, useState } from "react" import type { ReactNode } from "react" import type { ModelOption, ReviewerEvalConfig, ReviewerEvalScoreMode, ReviewerEvalSeverity, ReviewerEvalStartRequest, ReviewerEvalStatus, } from "@/lib/api" import { AppShell, SettingsSection } from "@/components/AppShell" import { Button } from "@/components/ui/button" import { Input } from "@/components/ui/input" import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue, } from "@/components/ui/select" import { Skeleton } from "@/components/ui/skeleton" import { api } from "@/lib/api" import { useSession } from "@/lib/session" import { cn } from "@/lib/utils" export const Route = createFileRoute("/admin_/evals")({ component: ReviewerEvalPage }) function ReviewerEvalPage() { const session = useSession() if (session.isLoading) { return (
) } if (!session.data) return if (!session.data.is_admin) return return ( ) } const DEFAULT_REVIEWER_EVAL_CONFIG: ReviewerEvalConfig = { dataset_name: "openswe-reviewer-v1", experiment_prefix: "openswe-review-confidence", max_concurrency: 5, langsmith_project: "open-swe-evals", langgraph_url: "", assistant_id: "reviewer", model_id: "google_genai:gemini-3.5-flash", reasoning_effort: "medium", score_mode: "all_findings", severity_threshold: "medium", cap: 4, } interface ReviewerEvalFormState { dataset_name: string experiment_prefix: string max_concurrency: string langsmith_project: string langgraph_url: string assistant_id: string model_id: string reasoning_effort: string score_mode: ReviewerEvalScoreMode severity_threshold: ReviewerEvalSeverity cap: string limit: string } function formFromConfig( config: ReviewerEvalConfig, limit: number | null = null ): ReviewerEvalFormState { return { dataset_name: config.dataset_name, experiment_prefix: config.experiment_prefix, max_concurrency: String(config.max_concurrency), langsmith_project: config.langsmith_project, langgraph_url: config.langgraph_url, assistant_id: config.assistant_id, model_id: config.model_id, reasoning_effort: config.reasoning_effort, score_mode: config.score_mode, severity_threshold: config.severity_threshold, cap: String(config.cap), limit: limit ? String(limit) : "", } } function parsePositiveInt(label: string, value: string): number { const n = Number(value.trim()) if (!Number.isInteger(n) || n <= 0) { throw new Error(`${label} must be a positive whole number`) } return n } function parseOptionalPositiveInt(label: string, value: string): number | null { return value.trim() ? parsePositiveInt(label, value) : null } function parseNonNegativeInt(label: string, value: string): number { const n = Number(value.trim()) if (!Number.isInteger(n) || n < 0) { throw new Error(`${label} must be a non-negative whole number`) } return n } function requireText(label: string, value: string): string { const text = value.trim() if (!text) throw new Error(`${label} is required`) return text } function FieldGroup({ label, description, children, }: { label: string description?: string children: ReactNode }) { return (
{label} {description && ( {description} )}
{children}
) } function Field({ label, className, children, }: { label: string className?: string children: ReactNode }) { return (
{label} {children}
) } function ReviewerEvalRunner() { const qc = useQueryClient() const [draft, setDraft] = useState(() => formFromConfig(DEFAULT_REVIEWER_EVAL_CONFIG) ) const [error, setError] = useState(null) const initialized = useRef(false) const status = useQuery({ queryKey: ["reviewerEval"], queryFn: api.getReviewerEval, refetchInterval: (query) => query.state.data?.status === "running" ? 5000 : false, }) const options = useQuery({ queryKey: ["options"], queryFn: api.options }) const data = status.data const running = data?.status === "running" const currentModel: ModelOption | undefined = options.data?.models.find((m) => m.id === draft.model_id) ?? options.data?.models[0] useEffect(() => { if (initialized.current || !data?.config_snapshot) return initialized.current = true setDraft(formFromConfig(data.config_snapshot, data.limit)) }, [data?.config_snapshot, data?.limit]) useEffect(() => { if (!currentModel) return if (currentModel.id !== draft.model_id) { setDraft((current) => ({ ...current, model_id: currentModel.id, reasoning_effort: currentModel.default_effort, })) return } if (!currentModel.efforts.includes(draft.reasoning_effort)) { setDraft((current) => ({ ...current, reasoning_effort: currentModel.default_effort, })) } }, [currentModel, draft.model_id, draft.reasoning_effort]) const setField = ( key: TKey, value: ReviewerEvalFormState[TKey] ) => { setDraft((current) => ({ ...current, [key]: value })) } const buildRequest = (): ReviewerEvalStartRequest => { return { dataset_name: requireText("Dataset", draft.dataset_name), experiment_prefix: requireText("Run name", draft.experiment_prefix), max_concurrency: parsePositiveInt("Max concurrency", draft.max_concurrency), langsmith_project: requireText("LangSmith project", draft.langsmith_project), langgraph_url: draft.langgraph_url.trim(), assistant_id: requireText("Assistant ID", draft.assistant_id), model_id: requireText("Model", draft.model_id), reasoning_effort: requireText("Effort", draft.reasoning_effort), score_mode: draft.score_mode, severity_threshold: draft.severity_threshold, cap: parseNonNegativeInt("Cap", draft.cap), limit: parseOptionalPositiveInt("Limit", draft.limit), } } const onSuccess = (next: ReviewerEvalStatus) => { qc.setQueryData(["reviewerEval"], next) setError(null) } const onError = (e: Error) => setError(e.message) const start = useMutation({ mutationFn: () => { return api.startReviewerEval(buildRequest()) }, onSuccess, onError, }) const cancel = useMutation({ mutationFn: () => api.cancelReviewerEval(), onSuccess, onError, }) return ( <>
setField("experiment_prefix", e.target.value)} /> setField("dataset_name", e.target.value)} /> setField("limit", e.target.value)} /> setField("max_concurrency", e.target.value)} />
setField("cap", e.target.value)} />
setField("langsmith_project", e.target.value)} /> setField("langgraph_url", e.target.value)} /> setField("assistant_id", e.target.value)} />
Start Only one reviewer eval can run at a time.
{running && ( )}
{error &&

{error}

}
) } function ReviewerEvalStatusView({ data }: { data: ReviewerEvalStatus | null }) { if (!data) { return (
Loading reviewer eval status…
) } const config = data.config_snapshot return (
{data.started_at && ( )} {data.finished_at && ( )} {data.experiment_url && ( View experiment in LangSmith )} {data.error && {data.error}}
) } function StatusLine({ label, value, strong = false, }: { label: string value: string | null | undefined strong?: boolean }) { return ( {label}:{" "} {value || "—"} ) } function ReviewerEvalLogs() { const status = useQuery({ queryKey: ["reviewerEval"], queryFn: api.getReviewerEval, refetchInterval: (query) => query.state.data?.status === "running" ? 5000 : false, }) const logTail = status.data?.log_tail ?? null const running = status.data?.status === "running" const scrollRef = useRef(null) const [follow, setFollow] = useState(true) const [copied, setCopied] = useState(false) useEffect(() => { if (follow && scrollRef.current) { scrollRef.current.scrollTop = scrollRef.current.scrollHeight } }, [logTail, follow]) const copyLogs = async () => { if (!logTail) return await navigator.clipboard.writeText(logTail) setCopied(true) window.setTimeout(() => setCopied(false), 1500) } return ( } >
{logTail ? (
 {
              const el = e.currentTarget
              const atBottom =
                el.scrollHeight - el.scrollTop - el.clientHeight < 24
              setFollow(atBottom)
            }}
            className="max-h-[28rem] overflow-auto whitespace-pre-wrap break-words rounded-md bg-muted/50 p-3 font-mono text-xs text-foreground"
          >
            {logTail}
          
) : (

{running ? "Waiting for output…" : "No output yet. Run an eval to see logs here."}

)}
) }