feat: handle images sent to non-vision models in Slack, Linear, and web UI (#1560)

* feat: handle images sent to non-vision models in Slack, Linear, and web UI

Add vision capability checks across all image input paths. When a user
sends images to a text-only model (e.g. GLM 5.2, DeepSeek V4 Pro), the
images are now skipped and a warning is injected into the prompt instead
of sending unsupported content to the model.

- Slack: resolve model at webhook time, skip image fetch + add warning
- Linear: same pattern as Slack
- Queued message middleware: read resolved model from thread metadata,
  strip images from queued payloads for text-only models
- Web UI: disable submit + show inline warning when images are attached
  to a non-vision model selection
- Shared: resolve_agent_model_id helper + vision_not_supported_warning

Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>

* test: mock resolve_agent_model_id in Slack mention test

The test_process_slack_mention_queues_active_thread_message test was
missing a mock for the new resolve_agent_model_id call added to the
Slack webhook handler, causing a TypeError when image URLs triggered
the model resolution path.

Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>

* fix: include vision warning in queued payload for text-only models

Update the prompt variable (not just content_blocks) before clearing
image_urls so the queued payload also carries the warning text when a
Slack/Linear follow-up arrives while the thread is busy.

Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>

---------

Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
This commit is contained in:
Johannes du Plessis 2026-06-17 09:12:52 -07:00 • committed by GitHub
parent 46e8982b29
commit b19804536c
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
9 changed files with 220 additions and 19 deletions

View file

@ -10,6 +10,7 @@ from langgraph_sdk import get_client
from .options import SUPPORTED_MODEL_IDS, model_supports_effort, provider_fallback_pair
from .profiles import PROFILES_NAMESPACE
from .team_settings import get_team_default_model
from .user_mappings import cached_login_for_email, login_for_email
logger = logging.getLogger(__name__)
@ -135,3 +136,23 @@ def normalize_profile_subagent_overrides(
model_key="default_subagent_model",
effort_key="subagent_reasoning_effort",
)
async def resolve_agent_model_id(
github_login: str | None,
per_thread_model_id: str | None = None,
) -> str:
"""Resolve the agent model ID using the same precedence as ``get_agent``.
Order: per-thread override → profile override → team default.
"""
model_id, _effort = await get_team_default_model("agent")
if github_login:
profile = await load_profile(github_login)
if profile:
overridden_model, _ = normalize_profile_overrides(profile)
if overridden_model:
model_id = overridden_model
if isinstance(per_thread_model_id, str) and per_thread_model_id in SUPPORTED_MODEL_IDS:
model_id = per_thread_model_id
return model_id

View file

@ -14,8 +14,10 @@ import httpx
from langchain.agents.middleware import AgentState, before_model
from langgraph.config import get_config, get_store
from langgraph.runtime import Runtime
from langgraph_sdk import get_client
from ..utils.multimodal import fetch_image_block
from ..dashboard.options import model_supports_images
from ..utils.multimodal import fetch_image_block, vision_not_supported_warning
logger = logging.getLogger(__name__)
@ -34,8 +36,25 @@ class LinearNotifyState(AgentState):
linear_messages_sent_count: int
async def _resolve_thread_model_id(thread_id: str) -> str | None:
"""Read the resolved model from thread metadata (set by ``get_agent``)."""
try:
client = get_client()
thread = await client.threads.get(thread_id)
metadata = thread.get("metadata") if isinstance(thread, dict) else None
if not isinstance(metadata, dict):
return None
model = metadata.get("model")
return model if isinstance(model, str) and model else None
except Exception:
logger.debug("Could not read thread metadata for model resolution", exc_info=True)
return None
async def _build_blocks_from_payload(
payload: dict[str, Any],
*,
model_id: str | None = None,
) -> list[dict[str, Any]]:
text = payload.get("text", "")
image_urls = payload.get("image_urls", []) or []
@ -48,6 +67,18 @@ async def _build_blocks_from_payload(
if not image_urls:
return blocks
if model_id and not model_supports_images(model_id):
logger.warning(
"Skipping %d queued image(s): model %s does not support images",
len(image_urls),
model_id,
)
if text:
blocks[0] = {
"type": "text",
"text": text + vision_not_supported_warning(model_id, len(image_urls)),
}
return blocks
async with httpx.AsyncClient() as client:
for image_url in image_urls:
image_block = await fetch_image_block(image_url, client)
@ -117,6 +148,15 @@ async def check_message_queue_before_model( # noqa: PLR0911
thread_id,
)
has_images = any(
isinstance(msg.get("content"), dict)
and (msg["content"].get("image_urls") or msg["content"].get("images"))
for msg in queued_messages
)
resolved_model_id: str | None = None
if has_images:
resolved_model_id = await _resolve_thread_model_id(thread_id)
content_blocks: list[dict[str, Any]] = []
for msg in queued_messages:
content = msg.get("content")
@ -126,7 +166,7 @@ async def check_message_queue_before_model( # noqa: PLR0911
"text" in content or "image_urls" in content or "images" in content
):
logger.debug("Queued message contains text + image URLs")
blocks = await _build_blocks_from_payload(content)
blocks = await _build_blocks_from_payload(content, model_id=resolved_model_id)
content_blocks.extend(blocks)
continue
if isinstance(content, list):

View file

@ -37,6 +37,15 @@ def extract_image_urls(text: str) -> list[str]:
return deduped
def vision_not_supported_warning(model_id: str, image_count: int) -> str:
"""Build a prompt-visible warning when images are sent to a text-only model."""
return (
f"\n\n**Note:** {image_count} image(s) were attached but the current model "
f"({model_id}) does not support image input. The images were not included. "
"Please switch to a vision-enabled model to process images."
)
async def fetch_image_block(
image_url: str,
client: httpx.AsyncClient,

View file

@ -24,11 +24,13 @@ from .ci_autofix import handle_ci_failure, handle_review_feedback
from .dashboard import router as dashboard_router
from .dashboard.agent_overrides import (
get_profile_default_repo,
resolve_agent_model_id,
resolve_login_from_email_async,
)
from .dashboard.autofix_state import set_pr_autofix_disabled
from .dashboard.enabled_repos import is_review_repo_enabled
from .dashboard.oauth import build_settings_url
from .dashboard.options import model_supports_images
from .dashboard.profiles import get_profile, get_valid_access_token, has_access_token_record
from .dashboard.team_settings import (
get_team_default_repo,
@ -96,7 +98,12 @@ from .utils.github_token import (
)
from .utils.linear import post_linear_trace_comment
from .utils.linear_team_repo_map import LINEAR_TEAM_TO_REPO
from .utils.multimodal import dedupe_urls, extract_image_urls, fetch_image_block
from .utils.multimodal import (
dedupe_urls,
extract_image_urls,
fetch_image_block,
vision_not_supported_warning,
)
from .utils.repo import extract_repo_from_text
from .utils.slack import (
GitHubPrRef,
@ -837,15 +844,27 @@ async def process_linear_issue( # noqa: PLR0912, PLR0915
content_blocks: list[dict[str, Any]] = [create_text_block(prompt)]
if image_urls:
image_urls = dedupe_urls(image_urls)
logger.info("Preparing %d image(s) for multimodal content", len(image_urls))
logger.debug("Image URLs: %s", image_urls)
linear_login = await resolve_login_from_email_async(user_email) if user_email else None
resolved_model_id = await resolve_agent_model_id(linear_login)
if model_supports_images(resolved_model_id):
logger.info("Preparing %d image(s) for multimodal content", len(image_urls))
logger.debug("Image URLs: %s", image_urls)
async with httpx.AsyncClient() as client:
for image_url in image_urls:
image_block = await fetch_image_block(image_url, client)
if image_block:
content_blocks.append(image_block)
logger.info("Built %d content block(s) for prompt", len(content_blocks))
async with httpx.AsyncClient() as client:
for image_url in image_urls:
image_block = await fetch_image_block(image_url, client)
if image_block:
content_blocks.append(image_block)
logger.info("Built %d content block(s) for prompt", len(content_blocks))
else:
logger.warning(
"Skipping %d image(s) for Linear issue: model %s does not support images",
len(image_urls),
resolved_model_id,
)
prompt += vision_not_supported_warning(resolved_model_id, len(image_urls))
content_blocks[0] = create_text_block(prompt)
image_urls = []
linear_project_id = ""
linear_issue_number = ""
@ -1068,18 +1087,30 @@ async def process_slack_mention(event_data: dict[str, Any], repo_config: dict[st
]
+ image_urls_from_links
)
if image_urls:
logger.info("Preparing %d image(s) for Slack mention", len(image_urls))
async with httpx.AsyncClient() as http_client:
for image_url in image_urls:
image_block = await fetch_image_block(image_url, http_client)
if image_block:
content_blocks.append(image_block)
mapped_login = await login_for_slack_id(user_id)
if not mapped_login and user_email:
mapped_login = await login_for_email(user_email)
if image_urls:
resolved_model_id = await resolve_agent_model_id(mapped_login)
if model_supports_images(resolved_model_id):
logger.info("Preparing %d image(s) for Slack mention", len(image_urls))
async with httpx.AsyncClient() as http_client:
for image_url in image_urls:
image_block = await fetch_image_block(image_url, http_client)
if image_block:
content_blocks.append(image_block)
else:
logger.warning(
"Skipping %d image(s) for Slack mention: model %s does not support images",
len(image_urls),
resolved_model_id,
)
prompt += vision_not_supported_warning(resolved_model_id, len(image_urls))
content_blocks[0] = create_text_block(prompt)
image_urls = []
# Open SWE opens PRs as the triggering user, so a run only proceeds when we
# have a valid user GitHub token. Users who have never signed in with
# GitHub, and users whose stored authorization is no longer usable, are

View file

@ -7,6 +7,7 @@ import pytest
from agent.middleware.check_message_queue import (
DASHBOARD_HANDOFF_MARKER,
_build_blocks_from_payload,
check_message_queue_before_model,
)
@ -53,3 +54,31 @@ async def test_check_message_queue_injects_dashboard_handoff_instruction() -> No
assert DASHBOARD_HANDOFF_MARKER in message["content"][0]["text"]
assert message["content"][1] == {"type": "text", "text": "continue in web"}
assert store.deleted == [(("queue", "thread-1"), "pending_messages")]
@pytest.mark.asyncio
async def test_build_blocks_skips_images_for_text_only_model() -> None:
payload = {
"text": "see this screenshot",
"image_urls": ["https://files.slack.com/fake.png"],
}
blocks = await _build_blocks_from_payload(
payload, model_id="fireworks:accounts/fireworks/models/glm-5p2"
)
assert len(blocks) == 1
assert blocks[0]["type"] == "text"
assert "does not support image input" in blocks[0]["text"]
@pytest.mark.asyncio
async def test_build_blocks_includes_images_for_vision_model() -> None:
payload: dict[str, Any] = {"text": "see this", "image_urls": []}
blocks = await _build_blocks_from_payload(payload, model_id="openai:gpt-5.5")
assert blocks == [{"type": "text", "text": "see this"}]
@pytest.mark.asyncio
async def test_build_blocks_no_model_check_fetches_images() -> None:
payload: dict[str, Any] = {"text": "see this", "image_urls": []}
blocks = await _build_blocks_from_payload(payload)
assert blocks == [{"type": "text", "text": "see this"}]

View file

@ -4,6 +4,7 @@ import pytest
from fastapi import HTTPException
from agent.dashboard import thread_api
from agent.dashboard.agent_overrides import resolve_agent_model_id
from agent.dashboard.options import model_supports_images
_TEXT_ONLY_MODEL = "fireworks:accounts/fireworks/models/deepseek-v4-pro"
@ -79,6 +80,43 @@ async def test_resolve_agent_model_choice_applies_request_before_profile(monkeyp
assert (model_id, effort) == ("anthropic:claude-opus-4-8", "high")
async def test_resolve_agent_model_id_defaults_to_team_default(monkeypatch) -> None:
async def fake_team_default(role: str) -> tuple[str, str]:
return _TEXT_ONLY_MODEL, "high"
monkeypatch.setattr("agent.dashboard.agent_overrides.get_team_default_model", fake_team_default)
monkeypatch.setattr("agent.dashboard.agent_overrides.load_profile", lambda login: None)
model_id = await resolve_agent_model_id(None)
assert model_id == _TEXT_ONLY_MODEL
async def test_resolve_agent_model_id_applies_profile_override(monkeypatch) -> None:
async def fake_team_default(role: str) -> tuple[str, str]:
return _TEXT_ONLY_MODEL, "high"
monkeypatch.setattr("agent.dashboard.agent_overrides.get_team_default_model", fake_team_default)
async def fake_load_profile(login: str) -> dict:
return {"default_model": _VISION_MODEL, "reasoning_effort": "medium"}
monkeypatch.setattr("agent.dashboard.agent_overrides.load_profile", fake_load_profile)
model_id = await resolve_agent_model_id("someuser")
assert model_id == _VISION_MODEL
async def test_resolve_agent_model_id_applies_per_thread_override(monkeypatch) -> None:
async def fake_team_default(role: str) -> tuple[str, str]:
return _TEXT_ONLY_MODEL, "high"
monkeypatch.setattr("agent.dashboard.agent_overrides.get_team_default_model", fake_team_default)
monkeypatch.setattr("agent.dashboard.agent_overrides.load_profile", lambda login: None)
model_id = await resolve_agent_model_id(None, per_thread_model_id="anthropic:claude-opus-4-8")
assert model_id == "anthropic:claude-opus-4-8"
def _new_thread_client(created: dict[str, object]) -> object:
class FakeThreads:
async def create(

View file

@ -1,6 +1,6 @@
from __future__ import annotations
from agent.utils.multimodal import extract_image_urls
from agent.utils.multimodal import extract_image_urls, vision_not_supported_warning
def test_extract_image_urls_empty() -> None:
@ -96,3 +96,15 @@ def test_extract_image_urls_mixed_markdown_and_direct() -> None:
"https://example.com/another.gif",
}
assert len(result) == 3
def test_vision_not_supported_warning_includes_model_and_count() -> None:
warning = vision_not_supported_warning("fireworks:.../glm-5p2", 2)
assert "glm-5p2" in warning
assert "2 image(s)" in warning
assert "does not support image input" in warning
def test_vision_not_supported_warning_singular() -> None:
warning = vision_not_supported_warning("fireworks:.../glm-5p2", 1)
assert "1 image(s)" in warning

View file

@ -709,6 +709,11 @@ def test_process_slack_mention_queues_active_thread_message(
monkeypatch.setattr(webapp, "refresh_user_mapping_cache", fake_refresh_cache)
monkeypatch.setattr(webapp, "get_valid_access_token", fake_get_valid_access_token)
async def fake_resolve_agent_model_id(github_login, per_thread_model_id=None):
return "openai:gpt-5.5"
monkeypatch.setattr(webapp, "resolve_agent_model_id", fake_resolve_agent_model_id)
thread_ts = "1700000000.000100"
event_ts = "1700000000.000200"
expected_thread_id = generate_thread_id_from_slack_thread("C123", thread_ts)

View file

@ -187,9 +187,17 @@ export const CloudPromptBar = memo(function CloudPromptBarComponent({
const selectionLabel = formatModelSelection(models, selection)
const selectedModelSupportsImages = useMemo(() => {
if (!selection || pendingImages.length === 0) return true
return models.some(
(m) => m.id === selection.modelId && m.supports_images
)
}, [selection, pendingImages.length, models])
const canSubmit =
!disabled &&
!isSubmitting &&
selectedModelSupportsImages &&
(value.trim().length > 0 || pendingImages.length > 0)
const handleSubmit = useCallback(async () => {
@ -389,6 +397,14 @@ export const CloudPromptBar = memo(function CloudPromptBarComponent({
</div>
)}
{!selectedModelSupportsImages && (
<div className="mb-2 rounded-md border border-[var(--ui-border)] bg-[var(--ui-panel-2)] px-3 py-1.5 text-xs text-[color:var(--ui-text-muted)]">
The selected model does not support image input. Remove the
image{pendingImages.length > 1 ? "s" : ""} or switch to a
vision-enabled model to send.
</div>
)}
<textarea
ref={inputRef}
rows={1}