This repository has been archived on 2026-08-04. You can view files and clone it, but cannot push or open issues or pull requests.
orchestrator/scripts/weekly_summary.py
Adam Moussa f05a98dba1 Add JSONL telemetry and weekly summary (Phase 4)
Observability slim layer. Every full run appends one JSON line to
~/.claude/logs/orchestrator/YYYY-MM-DD.jsonl with timestamp, sha256-prefix
task hash (raw task is never logged), retrieved memory names, router
choice, runtime, tokens in/out, success/error. risk_class and confidence
fields are reserved nulls for Phase 5.

- telemetry.py: log_run(), build_record(), task_hash(), token-usage
  extraction from AIMessage.usage_metadata. log_run swallows all
  exceptions — telemetry never kills a run.
- run.py: wraps app.invoke in try/except with a monotonic-clock window;
  logs on both success and failure. --route-only path is left unlogged
  (no agent work, doesn't represent a "run").
- scripts/weekly_summary.py: scans the last 7 days of JSONL and prints
  a markdown digest (routes, unknown rate, cross-review rate, success
  rate, total spend, mean tokens/route). Schedule via /schedule and
  pipe stdout to Slack from the scheduler.

Cost rates per route are rough Sonnet/Haiku/GPT/Gemini/DeepSeek defaults
suitable for spotting runaway prompts, not finance. Router tokens for
structured-output calls aren't captured (they don't surface through the
message trail); agent tokens are the dominant component anyway.

Validated: golden-set 21/21 still passing; full-run smoke writes
expected fields; weekly_summary.py prints clean markdown from a 2-run
log.
2026-05-15 11:49:05 -04:00

121 lines
4 KiB
Python
Executable file

#!/usr/bin/env python3
"""Weekly orchestrator summary.
Scans the last 7 days of telemetry JSONL under ~/.claude/logs/orchestrator/
and prints a markdown digest to stdout. Schedule via cron or /schedule;
the scheduler pipes the output to Slack.
Cost estimates use a route -> model-rate map; runs that hit done/unknown
are billed at Sonnet (router-only) rates. Numbers are rough — for spotting
runaway prompts, not finance.
"""
from __future__ import annotations
import datetime
import json
from collections import Counter, defaultdict
from pathlib import Path
LOG_DIR = Path("~/.claude/logs/orchestrator").expanduser()
WINDOW_DAYS = 7
# Rough $/MTok (input, output) by route. Sonnet for router-only routes.
COST_RATES: dict[str | None, tuple[float, float]] = {
"implementer": (3.00, 15.00),
"reviewer": (3.00, 15.00),
"researcher": (1.00, 5.00),
"cross_reviewer": (2.00, 8.00),
"scanner": (1.25, 5.00),
"fast_coder": (0.14, 0.28),
"connector": (3.00, 15.00),
"done": (3.00, 15.00),
"unknown": (3.00, 15.00),
None: (3.00, 15.00),
}
def load_recent_records(window_days: int = WINDOW_DAYS) -> list[dict]:
today = datetime.date.today()
out: list[dict] = []
for i in range(window_days):
day = today - datetime.timedelta(days=i)
path = LOG_DIR / f"{day.isoformat()}.jsonl"
if not path.exists():
continue
for line in path.read_text().splitlines():
line = line.strip()
if not line:
continue
try:
out.append(json.loads(line))
except json.JSONDecodeError:
continue
return out
def estimate_cost(record: dict) -> float:
rate_in, rate_out = COST_RATES.get(record.get("route"), COST_RATES[None])
tokens_in = record.get("tokens_in", 0) or 0
tokens_out = record.get("tokens_out", 0) or 0
return (tokens_in * rate_in + tokens_out * rate_out) / 1_000_000
def summarize(records: list[dict], window_days: int = WINDOW_DAYS) -> str:
if not records:
return (
"# Orchestrator weekly summary\n\n"
f"_No runs logged in the last {window_days} days._"
)
total = len(records)
successes = sum(1 for r in records if r.get("success"))
routes = Counter(r.get("route") for r in records)
by_route_tokens: dict[str, list[tuple[int, int]]] = defaultdict(list)
for r in records:
by_route_tokens[r.get("route") or "null"].append(
(r.get("tokens_in", 0) or 0, r.get("tokens_out", 0) or 0)
)
total_cost = sum(estimate_cost(r) for r in records if r.get("success"))
total_tokens_in = sum((r.get("tokens_in", 0) or 0) for r in records)
total_tokens_out = sum((r.get("tokens_out", 0) or 0) for r in records)
unknown_rate = routes.get("unknown", 0) / total
cross_rate = routes.get("cross_reviewer", 0) / total
success_rate = successes / total
lines = [
"# Orchestrator weekly summary",
f"_Last {window_days} days · {total} runs · {successes} succeeded_",
"",
"## Routes",
]
for route, count in routes.most_common():
label = route if route is not None else "null"
lines.append(f"- `{label}`: {count} ({count / total:.0%})")
lines += [
"",
"## Health",
f"- Unknown route rate: **{unknown_rate:.1%}** (target <2%)",
f"- Cross-review rate: **{cross_rate:.1%}**",
f"- Success rate: **{success_rate:.1%}**",
"",
"## Spend (rough)",
f"- Total tokens: {total_tokens_in:,} in / {total_tokens_out:,} out",
f"- Estimated cost: **${total_cost:.2f}**",
"",
"## Tokens per route (mean in/out per run)",
]
for route, tokens in sorted(by_route_tokens.items(), key=lambda x: -len(x[1])):
n = len(tokens)
mean_in = sum(t[0] for t in tokens) / n
mean_out = sum(t[1] for t in tokens) / n
lines.append(f"- `{route}`: {mean_in:,.0f} in / {mean_out:,.0f} out ({n} runs)")
return "\n".join(lines)
if __name__ == "__main__":
print(summarize(load_recent_records()))