Follow-up to the Phases 1-2-4 modernization PR. - models.py: extend retry coverage to Gemini transient errors (google.api_core: ResourceExhausted, InternalServerError, ServiceUnavailable, DeadlineExceeded, GatewayTimeout). Before this, a 429 or 5xx from Google would crash the scanner route. Retriable count went from 6 to 11. - retriever.py: batch cache-miss embeddings into a single embed_documents() call instead of one embed_query() per memory. Cold rebuild of 87 memories went from ~30s to ~10s; one HTTPS round-trip instead of 87. - telemetry.py + scripts/weekly_summary.py: use UTC date for log filenames so they align with the UTC `timestamp` field inside each record. Eliminates the off-by-one near local midnight and matches Sea Haven's UTC-for-logs convention. - scripts/weekly_summary.py: include failed runs in the total cost estimate (they consumed tokens too) and surface a `(incl. $X on failed runs)` breakdown so outages are visible in the digest. Validated: ruff clean; weekly_summary.py still prints; retriever cold rebuild + cache hit both verified end-to-end.
104 lines
2.7 KiB
Python
104 lines
2.7 KiB
Python
import os
|
|
|
|
from langchain_anthropic import ChatAnthropic
|
|
from langchain_core.runnables import Runnable
|
|
from langchain_google_genai import ChatGoogleGenerativeAI
|
|
from langchain_openai import ChatOpenAI
|
|
|
|
# Model IDs — single source of truth. Bump here when families ship new revs.
|
|
CLAUDE_SONNET = "claude-sonnet-4-20250514"
|
|
CLAUDE_HAIKU = "claude-haiku-4-5-20251001"
|
|
OPENAI_CROSS_REVIEWER = "gpt-4.1"
|
|
GEMINI_SCANNER = "gemini-2.5-pro"
|
|
DEEPSEEK_FAST_CODER = "deepseek-coder"
|
|
DEEPSEEK_BASE_URL = "https://api.deepseek.com/v1"
|
|
|
|
|
|
def _collect_retriable_exceptions() -> tuple[type[BaseException], ...]:
|
|
excs: list[type[BaseException]] = []
|
|
try:
|
|
import anthropic
|
|
|
|
excs.extend(
|
|
[
|
|
anthropic.APIConnectionError,
|
|
anthropic.RateLimitError,
|
|
anthropic.InternalServerError,
|
|
]
|
|
)
|
|
except ImportError:
|
|
pass
|
|
try:
|
|
import openai
|
|
|
|
excs.extend(
|
|
[
|
|
openai.APIConnectionError,
|
|
openai.RateLimitError,
|
|
openai.InternalServerError,
|
|
]
|
|
)
|
|
except ImportError:
|
|
pass
|
|
try:
|
|
from google.api_core import exceptions as google_exceptions
|
|
|
|
excs.extend(
|
|
[
|
|
google_exceptions.ResourceExhausted,
|
|
google_exceptions.InternalServerError,
|
|
google_exceptions.ServiceUnavailable,
|
|
google_exceptions.DeadlineExceeded,
|
|
google_exceptions.GatewayTimeout,
|
|
]
|
|
)
|
|
except ImportError:
|
|
pass
|
|
return tuple(excs)
|
|
|
|
|
|
RETRIABLE_EXCEPTIONS = _collect_retriable_exceptions()
|
|
|
|
|
|
def with_retries(runnable: Runnable) -> Runnable:
|
|
"""Wrap an LLM runnable with up to 2 retries on transient provider errors."""
|
|
if not RETRIABLE_EXCEPTIONS:
|
|
return runnable
|
|
return runnable.with_retry(
|
|
retry_if_exception_type=RETRIABLE_EXCEPTIONS,
|
|
stop_after_attempt=3,
|
|
wait_exponential_jitter=True,
|
|
)
|
|
|
|
|
|
def get_orchestrator():
|
|
return ChatAnthropic(model=CLAUDE_SONNET, temperature=0)
|
|
|
|
|
|
def get_implementer():
|
|
return ChatAnthropic(model=CLAUDE_SONNET, temperature=0)
|
|
|
|
|
|
def get_reviewer():
|
|
return ChatAnthropic(model=CLAUDE_SONNET, temperature=0)
|
|
|
|
|
|
def get_researcher():
|
|
return ChatAnthropic(model=CLAUDE_HAIKU, temperature=0)
|
|
|
|
|
|
def get_cross_reviewer():
|
|
return ChatOpenAI(model=OPENAI_CROSS_REVIEWER, temperature=0.2)
|
|
|
|
|
|
def get_scanner():
|
|
return ChatGoogleGenerativeAI(model=GEMINI_SCANNER, temperature=0)
|
|
|
|
|
|
def get_fast_coder():
|
|
return ChatOpenAI(
|
|
model=DEEPSEEK_FAST_CODER,
|
|
base_url=DEEPSEEK_BASE_URL,
|
|
api_key=os.getenv("DEEPSEEK_API_KEY"),
|
|
temperature=0,
|
|
)
|