open-swe/agent/utils/thread_ops.py
Johannes du Plessis 98b824bd54
feat: add size caps for PR diff, fetch_url, Slack threads, pagination, message queue [closes OPE-51] (#1567)
* feat: add size caps for PR diff, fetch_url, Slack threads, pagination, message queue

Per-source byte/token caps with explicit truncation markers to prevent
unbounded payloads from blowing up LLM context/memory.

- reviewer_diff.py: cap PR diff at 200K chars with head+tail truncation
- fetch_url.py: cap markdownify output at 100K chars
- slack.py: cap thread message fetch at 500 messages
- github_comments.py: cap _fetch_paginated at 50 pages
- thread_ops.py: cap queued messages at 100 (drop oldest)

Closes OPE-51

Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>

* fix: compute diff line set from full diff, keep most recent Slack messages

Address PR review comments:

1. Truncated diffs rejected valid findings: fetch_pr_diff now returns the
   full diff; truncate_diff is called separately in reviewer.py so the
   line set used for add_finding/publish_review validation is computed
   from the complete diff, not the truncated prompt text.

2. Slack cap dropped recent thread context: fetch_slack_thread_messages
   now keeps the most recent SLACK_THREAD_MAX_MESSAGES messages (was
   keeping the oldest). The tool surfaces a truncation marker in the
   formatted output so the LLM knows the thread was truncated.

Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>

---------

Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
2026-06-17 15:40:59 -07:00

78 lines
2.7 KiB
Python

"""Shared LangGraph thread helpers for webhooks and the dashboard."""
from __future__ import annotations
import logging
import os
from typing import Any
from langgraph_sdk import get_client
logger = logging.getLogger(__name__)
MAX_QUEUED_MESSAGES = 100
def langgraph_url() -> str:
return os.environ.get("LANGGRAPH_URL") or os.environ.get(
"LANGGRAPH_URL_PROD", "http://localhost:2024"
)
def langgraph_client():
return get_client(url=langgraph_url())
async def get_thread_active_status(thread_id: str) -> bool | None:
"""Return whether the thread is active, or None when status cannot be determined."""
try:
thread = await langgraph_client().threads.get(thread_id)
status = thread.get("status", "idle") if isinstance(thread, dict) else "idle"
logger.info("Thread %s status check: status=%s", thread_id, status)
return status == "busy"
except Exception as exc: # noqa: BLE001
logger.warning("Failed to get thread status for %s: %s", thread_id, exc)
return None
async def is_thread_active(thread_id: str) -> bool:
"""Return whether the thread currently has a running run."""
return await get_thread_active_status(thread_id) is True
async def queue_message_for_thread(
thread_id: str, message_content: str | list[dict[str, Any]] | dict[str, Any]
) -> bool:
"""Queue a follow-up message for a busy thread (FIFO store namespace)."""
client = langgraph_client()
try:
namespace = ("queue", thread_id)
key = "pending_messages"
new_message = {"content": message_content}
existing_messages: list[dict[str, Any]] = []
try:
existing_item = await client.store.get_item(namespace, key)
if existing_item and existing_item.get("value"):
existing_messages = existing_item["value"].get("messages", [])
except Exception: # noqa: BLE001
logger.debug("No existing queued messages for thread %s", thread_id)
existing_messages.append(new_message)
if len(existing_messages) > MAX_QUEUED_MESSAGES:
existing_messages = existing_messages[-MAX_QUEUED_MESSAGES:]
logger.warning(
"Thread %s queue capped at %d messages (dropped oldest)",
thread_id,
MAX_QUEUED_MESSAGES,
)
await client.store.put_item(namespace, key, {"messages": existing_messages})
logger.info(
"Queued message for thread %s (total queued: %d)",
thread_id,
len(existing_messages),
)
return True
except Exception:
logger.exception("Failed to queue message for thread %s", thread_id)
return False