open-swe/tests/test_stale_sandbox_creating.py

102 lines
3.9 KiB
Python
Raw Normal View History

feat: outcomes dataset + bootstrap/continual split via skills (#1365) * fix: reset stale sandbox creation sentinel Co-authored-by: Johannes du Plessis <51395795+johannes117@users.noreply.github.com> * fix: treat SANDBOX_CREATING as a timestamped cross-process lock Only reset the sentinel when proven stale (older than the creation timeout); otherwise wait for the worker that holds the lock so a concurrent run does not create a duplicate sandbox. * feat(analyzer): outcomes dataset + bootstrap/continual split via skills Rename the review_style_analyzer graph to `analyzer` and split it into two modes, plus capture reviewer finding outcomes for continual learning. - Outcomes dataset: upsert resolved-by-commit (positive), dismissed (false positive), and GitHub/Slack thumbs findings into a single LangSmith dataset (openswe-reviewer-outcomes), keyed deterministically per finding+source. Emit points wired into update_finding, resolve_finding_thread, and the GitHub/Slack reaction handlers. - Two playbooks delivered as deepagents skills (bootstrap-repo-analysis, continual-learning), served as virtual files via a CompositeBackend /skills/ route + StateBackend (seeded into the run files channel at invoke time, never written to the sandbox). Mode is set by the launcher; continual runs fall back to the GitHub App installation token. - Split launcher into start_bootstrap_analysis + start_continual_run; register a per-repo nightly continual-learning cron when bootstrap completes. - New read_finding_outcomes tool feeds confirmed/dismissed findings back to the continual playbook. Tests for outcome label mapping, skills helper, and cron idempotency. * fix(analyzer): anchor continual cron runs to a real thread_id The nightly continual-learning cron is threadless, and get_analyzer early-returns an empty agent when configurable.thread_id is missing — so every cron-launched run no-op'd before reading outcomes or saving a refined prompt. Include the repo's deterministic analyzer thread_id in the continual run configurable so the run executes; the threadless run carries no message history, so nightly runs don't accumulate context. * refactor(analyzer): move cron lifecycle calls out of the review-styles store Drop the inline `analyzer_cron` imports from review_styles.py (added only to dodge a circular import) by relocating the cron-trigger calls to the layer above the store: registration to the save_review_style tool (after a prompt is saved) and removal to the dashboard delete route. review_styles.py is now a pure store again with top-level imports only. * refactor: hoist reviewer_outcomes imports to module level Move the two inline emit_finding_status_outcome imports introduced in this PR (update_finding, resolve_finding_thread) to top-level imports. reviewer_outcomes only depends on langsmith, so there is no circular import to avoid. --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
2026-06-01 13:25:12 -07:00
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
from agent.server import (
SANDBOX_BACKENDS,
SANDBOX_CREATING,
SANDBOX_CREATION_TIMEOUT,
ensure_sandbox_for_thread,
)
@pytest.mark.asyncio
async def test_stale_sandbox_creating_without_cached_backend_resets_and_creates() -> None:
thread_id = "thread-stale-creating"
SANDBOX_BACKENDS.clear()
sandbox_backend = MagicMock()
sandbox_backend.id = "sandbox-new"
stale_at = 0.0 # epoch 0 → far older than the timeout
thread = {"metadata": {"sandbox_id": SANDBOX_CREATING, "sandbox_creating_at": stale_at}}
with (
patch(
"agent.server.get_sandbox_id_from_metadata",
new_callable=AsyncMock,
return_value=SANDBOX_CREATING,
),
patch("agent.server.client.threads.get", new_callable=AsyncMock, return_value=thread),
patch("agent.server._create_sandbox_with_proxy", new_callable=AsyncMock) as create_sandbox,
patch("agent.server._configure_git_identity", new_callable=AsyncMock),
patch("agent.server.client.threads.update", new_callable=AsyncMock) as update_thread,
):
create_sandbox.return_value = sandbox_backend
result = await ensure_sandbox_for_thread(thread_id)
assert result.id == "sandbox-new"
create_sandbox.assert_awaited_once()
# Stale sentinel reset (id + timestamp cleared), then fresh creation claims it.
assert update_thread.await_args_list[0].kwargs == {
"thread_id": thread_id,
"metadata": {"sandbox_id": None, "sandbox_creating_at": None},
}
assert update_thread.await_args_list[1].kwargs["metadata"]["sandbox_id"] == SANDBOX_CREATING
assert update_thread.await_args_list[2].kwargs == {
"thread_id": thread_id,
"metadata": {"sandbox_id": "sandbox-new"},
}
SANDBOX_BACKENDS.clear()
@pytest.mark.asyncio
async def test_fresh_sandbox_creating_waits_for_other_worker() -> None:
"""A recent sentinel from another worker must be waited on, not overwritten."""
thread_id = "thread-concurrent-creating"
SANDBOX_BACKENDS.clear()
existing_backend = MagicMock()
existing_backend.id = "sandbox-existing"
import time
fresh_at = time.time() # well within the timeout
threads = [
{"metadata": {"sandbox_id": SANDBOX_CREATING, "sandbox_creating_at": fresh_at}},
{"metadata": {"sandbox_id": "sandbox-existing", "sandbox_creating_at": fresh_at}},
]
async def passthrough(
sb, _thread_id, _github_proxy_token=None, _github_proxy_repositories=None
):
feat: outcomes dataset + bootstrap/continual split via skills (#1365) * fix: reset stale sandbox creation sentinel Co-authored-by: Johannes du Plessis <51395795+johannes117@users.noreply.github.com> * fix: treat SANDBOX_CREATING as a timestamped cross-process lock Only reset the sentinel when proven stale (older than the creation timeout); otherwise wait for the worker that holds the lock so a concurrent run does not create a duplicate sandbox. * feat(analyzer): outcomes dataset + bootstrap/continual split via skills Rename the review_style_analyzer graph to `analyzer` and split it into two modes, plus capture reviewer finding outcomes for continual learning. - Outcomes dataset: upsert resolved-by-commit (positive), dismissed (false positive), and GitHub/Slack thumbs findings into a single LangSmith dataset (openswe-reviewer-outcomes), keyed deterministically per finding+source. Emit points wired into update_finding, resolve_finding_thread, and the GitHub/Slack reaction handlers. - Two playbooks delivered as deepagents skills (bootstrap-repo-analysis, continual-learning), served as virtual files via a CompositeBackend /skills/ route + StateBackend (seeded into the run files channel at invoke time, never written to the sandbox). Mode is set by the launcher; continual runs fall back to the GitHub App installation token. - Split launcher into start_bootstrap_analysis + start_continual_run; register a per-repo nightly continual-learning cron when bootstrap completes. - New read_finding_outcomes tool feeds confirmed/dismissed findings back to the continual playbook. Tests for outcome label mapping, skills helper, and cron idempotency. * fix(analyzer): anchor continual cron runs to a real thread_id The nightly continual-learning cron is threadless, and get_analyzer early-returns an empty agent when configurable.thread_id is missing — so every cron-launched run no-op'd before reading outcomes or saving a refined prompt. Include the repo's deterministic analyzer thread_id in the continual run configurable so the run executes; the threadless run carries no message history, so nightly runs don't accumulate context. * refactor(analyzer): move cron lifecycle calls out of the review-styles store Drop the inline `analyzer_cron` imports from review_styles.py (added only to dodge a circular import) by relocating the cron-trigger calls to the layer above the store: registration to the save_review_style tool (after a prompt is saved) and removal to the dashboard delete route. review_styles.py is now a pure store again with top-level imports only. * refactor: hoist reviewer_outcomes imports to module level Move the two inline emit_finding_status_outcome imports introduced in this PR (update_finding, resolve_finding_thread) to top-level imports. reviewer_outcomes only depends on langsmith, so there is no circular import to avoid. --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
2026-06-01 13:25:12 -07:00
return sb
with (
patch(
"agent.server.get_sandbox_id_from_metadata",
new_callable=AsyncMock,
return_value=SANDBOX_CREATING,
),
patch("agent.server.client.threads.get", new_callable=AsyncMock, side_effect=threads),
patch("agent.server.asyncio.sleep", new_callable=AsyncMock),
patch("agent.server.create_sandbox", return_value=existing_backend) as connect_sandbox,
patch("agent.server._create_sandbox_with_proxy", new_callable=AsyncMock) as create_sandbox,
patch("agent.server.check_or_recreate_sandbox", side_effect=passthrough),
patch("agent.server._refresh_github_proxy_or_recreate", side_effect=passthrough),
patch("agent.server._configure_git_identity", new_callable=AsyncMock),
patch("agent.server.client.threads.update", new_callable=AsyncMock) as update_thread,
):
result = await ensure_sandbox_for_thread(thread_id)
assert result.id == "sandbox-existing"
# Connected to the worker's sandbox; no second sandbox was created.
connect_sandbox.assert_called_once_with("sandbox-existing")
create_sandbox.assert_not_awaited()
# The fresh sentinel was never reset.
for call in update_thread.await_args_list:
assert call.kwargs["metadata"] != {"sandbox_id": None, "sandbox_creating_at": None}
assert SANDBOX_CREATION_TIMEOUT > 0
SANDBOX_BACKENDS.clear()