open-swe/tests/test_fireworks_model.py
seahaven-openswe[bot] eb98ff4c30
Some checks are pending
CI / Lint (push) Waiting to run
CI / Format check (push) Waiting to run
CI / Unit tests (push) Waiting to run
CI / Playwright E2E (push) Waiting to run
feat: add 3 verified Fireworks models to selectable set (#79)
* Add 10 Fireworks models to selectable set

Surface additional Fireworks-served models in the profile editor so
they can be chosen per-thread, per-profile, and as team defaults. Each
entry carries its recommended efforts and image support; only MiniMax
M3 is multimodal.

Refs: #78

* Suppress reasoning_effort on non-reasoning models

Instruct-only Fireworks ids (kimi-k2-instruct-0905, mistral-large-3-fp8,
qwen3-30b-a3b-instruct-2507) don't reason, so sending reasoning_effort
either 400s (unusable at default effort) or is a silent no-op. Add a
per-model reasoning flag (default True) and omit the param entirely for
ids marked non-reasoning.

Refs: #78

* Gate out 7 undeployed Fireworks models

Account serverless probe returned 404 for 7 of the 10 proposed ids, so
only minimax-m3, gpt-oss-120b, and deepseek-v4-flash are callable. Keep
those 3 and drop the rest. All 3 survivors are reasoning-capable, so the
per-model reasoning-effort suppression added earlier is no longer needed
and is reverted.

Refs: #78

---------

Co-authored-by: amoussa1229 <166072409+amoussa1229@users.noreply.github.com>
2026-06-30 15:31:22 -04:00

134 lines
4.5 KiB
Python

import pytest
from agent.dashboard.options import (
SUPPORTED_MODEL_IDS,
SUPPORTED_MODELS,
model_supports_effort,
model_supports_images,
)
from agent.utils.model import (
fallback_model_id_for,
fireworks_reasoning_effort_for,
provider_model_kwargs,
)
_FIREWORKS_PREFIX = "fireworks:accounts/fireworks/models/"
NEW_FIREWORKS_MODELS = {
"minimax-m3": (["medium", "high"], "high", True),
"gpt-oss-120b": (["low", "medium", "high"], "medium", False),
"deepseek-v4-flash": (["none", "medium", "high"], "high", False),
}
_ALL_EFFORTS = ("none", "low", "medium", "high", "xhigh", "max")
def test_fireworks_reasoning_effort_maps_effort() -> None:
for effort in ("none", "low", "medium", "high", "xhigh", "max"):
assert fireworks_reasoning_effort_for(effort) == effort
assert fireworks_reasoning_effort_for("bogus") is None
assert fireworks_reasoning_effort_for(None) is None
def test_provider_model_kwargs_for_fireworks() -> None:
kwargs = provider_model_kwargs(
"fireworks:accounts/fireworks/models/kimi-k2p7-code",
"high",
max_tokens=16_000,
)
assert kwargs["max_tokens"] == 16_000
assert kwargs["model_kwargs"] == {"reasoning_effort": "high"}
def test_kimi_k2p7_is_supported() -> None:
kimi_k2p7 = next(
(m for m in SUPPORTED_MODELS if m["id"].endswith("kimi-k2p7-code")),
None,
)
assert kimi_k2p7 is not None
assert kimi_k2p7["efforts"] == ["low", "medium", "high"]
assert "none" not in kimi_k2p7["efforts"]
assert kimi_k2p7["default_effort"] == "high"
kwargs = provider_model_kwargs(kimi_k2p7["id"], "high", max_tokens=16_000)
assert kwargs["model_kwargs"] == {"reasoning_effort": "high"}
def test_provider_model_kwargs_for_fireworks_none_disables_reasoning() -> None:
kwargs = provider_model_kwargs(
"fireworks:accounts/fireworks/models/deepseek-v4-pro",
"none",
max_tokens=16_000,
)
assert kwargs["model_kwargs"] == {"reasoning_effort": "none"}
def test_provider_model_kwargs_for_fireworks_unknown_effort_omits_reasoning() -> None:
kwargs = provider_model_kwargs(
"fireworks:accounts/fireworks/models/glm-5p1",
"bogus",
max_tokens=16_000,
)
assert "model_kwargs" not in kwargs
@pytest.mark.parametrize("slug", sorted(NEW_FIREWORKS_MODELS))
def test_new_fireworks_model_is_supported(slug: str) -> None:
model_id = _FIREWORKS_PREFIX + slug
assert model_id in SUPPORTED_MODEL_IDS
model = next(m for m in SUPPORTED_MODELS if m["id"] == model_id)
efforts, default_effort, supports_images = NEW_FIREWORKS_MODELS[slug]
assert model["efforts"] == efforts
assert model["default_effort"] == default_effort
assert model["supports_images"] is supports_images
@pytest.mark.parametrize("slug", sorted(NEW_FIREWORKS_MODELS))
def test_new_fireworks_model_supports_only_listed_efforts(slug: str) -> None:
model_id = _FIREWORKS_PREFIX + slug
efforts = NEW_FIREWORKS_MODELS[slug][0]
for effort in efforts:
assert model_supports_effort(model_id, effort) is True
for effort in _ALL_EFFORTS:
if effort not in efforts:
assert model_supports_effort(model_id, effort) is False
@pytest.mark.parametrize(
"slug",
[
"qwen3-coder-480b-a35b-instruct",
"kimi-k2-thinking",
"kimi-k2-instruct-0905",
"glm-4p6",
"mistral-large-3-fp8",
"deepseek-v3p2",
"qwen3-30b-a3b-instruct-2507",
],
)
def test_unavailable_fireworks_models_are_gated_out(slug: str) -> None:
assert _FIREWORKS_PREFIX + slug not in SUPPORTED_MODEL_IDS
@pytest.mark.parametrize("slug", sorted(NEW_FIREWORKS_MODELS))
def test_only_minimax_m3_supports_images(slug: str) -> None:
model_id = _FIREWORKS_PREFIX + slug
assert model_supports_images(model_id) is (slug == "minimax-m3")
def test_fireworks_falls_back_to_bedrock() -> None:
assert (
fallback_model_id_for("fireworks:accounts/fireworks/models/deepseek-v4-pro")
== "bedrock_converse:us.anthropic.claude-opus-4-8"
)
@pytest.mark.parametrize(
("model_id", "effort"),
[(m["id"], effort) for m in SUPPORTED_MODELS for effort in m["efforts"]],
)
def test_every_supported_effort_translates_to_a_reasoning_kwarg(model_id: str, effort: str) -> None:
"""Each effort surfaced in the UI must map to a provider reasoning param."""
kwargs = provider_model_kwargs(model_id, effort, max_tokens=16_000)
assert set(kwargs) - {"max_tokens"}, (
f"{model_id} effort {effort!r} did not produce a reasoning kwarg"
)