mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 23:13:15 +00:00
* feat: add size caps for PR diff, fetch_url, Slack threads, pagination, message queue Per-source byte/token caps with explicit truncation markers to prevent unbounded payloads from blowing up LLM context/memory. - reviewer_diff.py: cap PR diff at 200K chars with head+tail truncation - fetch_url.py: cap markdownify output at 100K chars - slack.py: cap thread message fetch at 500 messages - github_comments.py: cap _fetch_paginated at 50 pages - thread_ops.py: cap queued messages at 100 (drop oldest) Closes OPE-51 Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> * fix: compute diff line set from full diff, keep most recent Slack messages Address PR review comments: 1. Truncated diffs rejected valid findings: fetch_pr_diff now returns the full diff; truncate_diff is called separately in reviewer.py so the line set used for add_finding/publish_review validation is computed from the complete diff, not the truncated prompt text. 2. Slack cap dropped recent thread context: fetch_slack_thread_messages now keeps the most recent SLACK_THREAD_MAX_MESSAGES messages (was keeping the oldest). The tool surfaces a truncation marker in the formatted output so the LLM knows the thread was truncated. Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
68 lines
2.4 KiB
Python
68 lines
2.4 KiB
Python
from typing import Any
|
|
|
|
import requests
|
|
from markdownify import markdownify
|
|
|
|
from .http_request import _request_with_safe_redirects
|
|
|
|
FETCH_URL_MAX_CHARS = 100_000
|
|
|
|
|
|
def fetch_url(url: str, timeout: int = 30) -> dict[str, Any]:
|
|
"""Fetch content from a URL and convert HTML to markdown format.
|
|
|
|
This tool fetches web page content and converts it to clean markdown text,
|
|
making it easy to read and process HTML content. After receiving the markdown,
|
|
you MUST synthesize the information into a natural, helpful response for the user.
|
|
|
|
Args:
|
|
url: The URL to fetch (must be a valid HTTP/HTTPS URL)
|
|
timeout: Request timeout in seconds (default: 30)
|
|
|
|
Returns:
|
|
Dictionary containing:
|
|
- success: Whether the request succeeded
|
|
- url: The final URL after redirects
|
|
- markdown_content: The page content converted to markdown
|
|
- status_code: HTTP status code
|
|
- content_length: Length of the markdown content in characters
|
|
|
|
IMPORTANT: After using this tool:
|
|
1. Read through the markdown content
|
|
2. Extract relevant information that answers the user's question
|
|
3. Synthesize this into a clear, natural language response
|
|
4. NEVER show the raw markdown to the user unless specifically requested
|
|
"""
|
|
try:
|
|
response, blocked = _request_with_safe_redirects(
|
|
"GET",
|
|
url,
|
|
timeout=timeout,
|
|
headers={"User-Agent": "Mozilla/5.0 (compatible; DeepAgents/1.0)"},
|
|
)
|
|
if blocked:
|
|
return {
|
|
"error": blocked["content"],
|
|
"status_code": blocked["status_code"],
|
|
"url": blocked["url"],
|
|
}
|
|
|
|
response.raise_for_status()
|
|
|
|
# Convert HTML content to markdown
|
|
markdown_content = markdownify(response.text)
|
|
|
|
if len(markdown_content) > FETCH_URL_MAX_CHARS:
|
|
markdown_content = (
|
|
markdown_content[:FETCH_URL_MAX_CHARS] + "\n... [content truncated: "
|
|
f"{FETCH_URL_MAX_CHARS}/{len(markdown_content)} chars]\n"
|
|
)
|
|
|
|
return {
|
|
"url": str(response.url),
|
|
"markdown_content": markdown_content,
|
|
"status_code": response.status_code,
|
|
"content_length": len(markdown_content),
|
|
}
|
|
except requests.exceptions.RequestException as e:
|
|
return {"error": f"Fetch URL error: {e!s}", "url": url}
|