diff --git a/README.md b/README.md index 9275b74..17f3e3d 100644 --- a/README.md +++ b/README.md @@ -210,8 +210,7 @@ Run these in the `llm-trunk` folder on the gateway's machine. | Command | What it does | |---|---| -| `python3 scripts/dashboard.py` | Live dashboard: savings, spend per request type, prompt-cache countdown per session, live feed. Starts empty; `--since 1h` adds history | -| `python3 scripts/watch.py` | Live log, one line per request | +| `python3 scripts/dashboard.py` | Live dashboard: savings, spend per request type, prompt-cache and sticky-tier countdowns per session, and a feed with one row per request (scroll with ↑/↓ or the mouse wheel, `g` for newest). Starts empty; `--since 1h` adds history | | `python3 scripts/report.py --since 24h` | Cost totals per request type, tier, day and session | | `python3 scripts/check_rules.py --since 1h` | Checks every logged request against the routing rules; exits with 1 on a violation | | `python3 scenarios/run.py --quick` | Drives a scripted Claude Code session through the gateway and checks each step's tier and answer (about $1–2 per run) | diff --git a/policy/litellm_callback.py b/policy/litellm_callback.py index 69a5f20..fb5d6c3 100644 --- a/policy/litellm_callback.py +++ b/policy/litellm_callback.py @@ -208,7 +208,7 @@ def _latest_user_blocks(data: dict) -> list[str]: def _log_event(kind: str, fields: dict) -> None: # One JSON line per outcome -- spend, deny, expired, failed -- which - # scripts/watch.py renders. Structured so a client-supplied value (e.g. a + # scripts/events.py parses. Structured so a client-supplied value (e.g. a # conversation id with spaces) can't break or hide a line. _log(f"llm-trunk {kind}: " + json.dumps(fields)) @@ -755,7 +755,7 @@ async def async_log_success_event(self, kwargs, response_obj, start_time, end_ti "input_tokens": _usage_value(usage, "prompt_tokens", "input_tokens"), "output_tokens": _usage_value(usage, "completion_tokens", "output_tokens"), # Reads only: a cache *write* is billed at 1.25x, the opposite - # of what the watcher's "(c)" tag claims. + # of what the dashboard's "(c)" tag claims. "cache_read_tokens": _usage_value(usage, "cache_read_input_tokens"), # Both are included in input_tokens (LiteLLM adds them in). "cache_write_tokens": _usage_value(usage, "cache_creation_input_tokens"), diff --git a/scenarios/run.py b/scenarios/run.py index c0300e7..6515520 100644 --- a/scenarios/run.py +++ b/scenarios/run.py @@ -46,7 +46,7 @@ import check_rules # noqa: E402 import dashboard # noqa: E402 from report import parse # noqa: E402 -from watch import pretty_model, tier_text # noqa: E402 +from events import pretty_model, tier_text # noqa: E402 GATEWAY = "http://127.0.0.1:4000" QUIET_SECONDS = 8 # a suggestion lands a few seconds after the reply diff --git a/scripts/check_rules.py b/scripts/check_rules.py index cbf089d..bb0392e 100644 --- a/scripts/check_rules.py +++ b/scripts/check_rules.py @@ -36,7 +36,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) from report import parse # noqa: E402 -from watch import REPO # noqa: E402 +from events import REPO # noqa: E402 from policy.decide import lowest, one_down # noqa: E402 from policy.models import model_key # noqa: E402 diff --git a/scripts/dashboard.py b/scripts/dashboard.py index 13c06b9..29f89fe 100644 --- a/scripts/dashboard.py +++ b/scripts/dashboard.py @@ -8,20 +8,26 @@ python3 scripts/dashboard.py --ttl 60m # prompt-cache lifetime (default 5m) python3 scripts/dashboard.py --window 2h # sessions shown if active this recently (default 30m) -Read-only, like scripts/watch.py, which stays the dependency-free view; this -one needs the `rich` package. "Without llm-trunk" is an estimate: each -request's own tokens priced at the model the client asked for (pricing.yaml), -so a request routed to a cheaper tier shows what the tier saved. Requests -logged before the gateway recorded the requested model can't be compared and -are counted separately. +Read-only; needs `rich` and `pyyaml` (pip install -r requirements-dev.txt). +Scroll the live feed with the arrow keys or the mouse wheel (or j/k), a page +with space/b, jump to the newest with g and the oldest with G; q quits. + +"Without llm-trunk" is an estimate: each request's own tokens priced at the +model the client asked for (pricing.yaml), so a request routed to a cheaper +tier shows what the tier saved. Requests logged before the gateway recorded +the requested model can't be compared and are counted separately. """ import argparse +import itertools +import os import queue import re +import select import subprocess import sys import threading import time +import zlib from collections import defaultdict, deque from datetime import datetime, timezone from pathlib import Path @@ -38,16 +44,20 @@ from report import parse # noqa: E402 -from policy.decide import REQUEST_TYPES # noqa: E402 (watch puts the repo on the path) +from events import ICONS, REPO, STAMP_RE, fmt_tokens, pretty_model, request_type_of, route_kind # noqa: E402 + +from policy.decide import REQUEST_TYPES # noqa: E402 (events puts the repo on the path) from policy.models import model_key # noqa: E402 -from watch import ICONS, REPO, STAMP_RE, fmt_tokens, pretty_model, request_type_of, route_kind # noqa: E402 MODEL_STYLES = {"Opus": "magenta", "Sonnet": "blue", "Haiku": "green", "Fable": "yellow"} +SESSION_STYLES = ["cyan", "yellow", "magenta", "green", "blue", "bright_cyan", "bright_yellow", "bright_magenta"] # Request types as the dashboard names them: Claude Code's own housekeeping # (titles, suggestions, away summaries, permission checks) reads as "internal". TYPE_LABELS = {"background": "internal"} -FEED_ROWS = 14 +FEED_HISTORY = 10_000 # rows kept for scrolling back SESSION_ROWS = 6 +HEADER_HEIGHT, MIDDLE_HEIGHT = 4, 12 +TYPES_WIDTH = 54 # the spend panel's widest line; the sessions panel gets the rest # --- Prices --------------------------------------------------------------------- @@ -119,7 +129,13 @@ def __init__(self, prices: dict, routes: dict, ttl_seconds: int, clock=time.time # request type -> [actual, without llm-trunk], over comparable requests only self.type_compare: dict[str, list[float]] = defaultdict(lambda: [0.0, 0.0]) self.sessions: dict[str, dict] = {} - self.feed: deque = deque(maxlen=FEED_ROWS) + # session -> (skill, epoch when its sticky route expires if left idle) + self.sticky: dict[str, tuple[str, float]] = {} + self.feed: deque = deque(maxlen=FEED_HISTORY) + # The feed lists newest first; scroll = how many newer rows are hidden + # above the view. page = rows the view showed last time it was drawn. + self.scroll = self.unseen = 0 + self.page = 1 def add(self, when: datetime, kind: str, event: dict) -> None: self.first = self.first or when @@ -128,7 +144,49 @@ def add(self, when: datetime, kind: str, event: dict) -> None: self.denies += 1 elif kind == "spend": saving = self._add_spend(when, event) + elif kind == "expired": + self.sticky.pop(event.get("session"), None) self.feed.appendleft((when, kind, event, saving)) + if self.scroll: + # Scrolled back: keep the rows in view where they are. + self.scroll = min(self.scroll + 1, self.max_scroll()) + self.unseen += 1 + + def max_scroll(self) -> int: + return max(0, len(self.feed) - self.page) + + def press(self, key: str) -> None: + """Scroll the feed: up/down a row, pgup/pgdn a page, home/end to either end.""" + steps = {"up": -1, "down": 1, "pgup": -self.page, "pgdn": self.page} + if key in steps: + self.scroll += steps[key] + elif key == "home": + self.scroll = 0 + elif key == "end": + self.scroll = self.max_scroll() + self.scroll = max(0, min(self.scroll, self.max_scroll())) + if not self.scroll: + self.unseen = 0 + + def _track_sticky(self, when: datetime, event: dict) -> None: + # Every request but a subagent's reports the session's sticky timer: + # its time left, or none once the route has ended. + session = event.get("session") + if not session or event.get("request_type") == "subagent": + return + left, skill = event.get("sticky_left_s"), event.get("skill_id") + if isinstance(left, (int, float)) and skill: + self.sticky[session] = (skill, when.timestamp() + left) + else: + self.sticky.pop(session, None) + + def sticky_left(self, session: str) -> tuple[str, int] | None: + """(skill, seconds left) while the session's sticky route is live.""" + if session not in self.sticky: + return None + skill, expires = self.sticky[session] + left = int(expires - self.clock()) + return (skill, left) if left > 0 else None def _add_spend(self, when: datetime, event: dict) -> float | None: logged = event.get("cost") @@ -140,6 +198,7 @@ def _add_spend(self, when: datetime, event: dict) -> float | None: self.type_cost[kind] += cost self.input_tokens += event.get("input_tokens") or 0 self.cached_tokens += event.get("cache_read_tokens") or 0 + self._track_sticky(when, event) # The cache clock and "next message" price follow the main conversation: # subagents, titles and permission checks share the session but run # another model on another context. @@ -259,66 +318,129 @@ def types_panel(dash: Dashboard) -> Panel: return Panel(table, title="spend by request type · vs model asked for", title_align="left") +def _session_text(session: str | None) -> Text: + if not session: + return Text("—", style="dim") + return Text(session, style=SESSION_STYLES[zlib.crc32(session.encode()) % len(SESSION_STYLES)]) + + +def _clock(seconds: int) -> str: + return f"{seconds // 60}:{seconds % 60:02d}" + + def sessions_panel(dash: Dashboard) -> Panel: - table = Table.grid(padding=(0, 1)) - for _ in range(4): - table.add_column(no_wrap=True, overflow="ellipsis") + table = Table(box=None, padding=(0, 1), show_edge=False, header_style="dim") + for name in ("SESSION", "MODEL", "CACHE", "NEXT MESSAGE"): + table.add_column(name, no_wrap=True, overflow="ellipsis") + # The one column that gives way on a narrow terminal (its text still won't wrap). + table.add_column("STICKY", no_wrap=False) now = dash.clock() active = [(session, state) for session, state in dash.sessions.items() if now - state["last"] <= dash.window] recent = sorted(active, key=lambda item: -item[1]["last"])[:SESSION_ROWS] for session, state in recent: warm, left, if_warm, if_cold = dash.cache_state(state) - status = Text(f"● {left // 60}:{left % 60:02d}", style="green") if warm else Text("○ cold", style="bold red") + status = Text(f"● {_clock(left)}", style="green") if warm else Text("○ cold", style="bold red") if if_warm is None or if_cold is None: next_message = Text("") elif warm: - next_message = Text(f"next {_money(if_warm)} · cold {_money(if_cold)}", style="dim") + next_message = Text(f"{_money(if_warm)} · cold {_money(if_cold)}", style="dim") else: next_message = Text(f"next re-cache costs {_money(if_cold)}", style="red") - table.add_row(Text(session, style="bold"), _model_text(state["model"]), status, next_message) + sticky = dash.sticky_left(session) + sticky_text = Text(f"{ICONS['sticky']} {_clock(sticky[1])} {sticky[0]}" if sticky else "—", + style="yellow" if sticky else "dim", no_wrap=True, overflow="ellipsis") + table.add_row(_session_text(session), _model_text(state["model"]), status, next_message, sticky_text) if not recent: table.add_row(Text(f"no activity in the last {_duration(dash.window)}", style="dim")) title = f"sessions active in the last {_duration(dash.window)} · cache {_duration(dash.ttl)}" return Panel(table, title=title, title_align="left") -def feed_panel(dash: Dashboard) -> Panel: - table = Table(box=None, padding=(0, 1), show_edge=False, header_style="dim") - for name, justify in (("TIME", "left"), ("ROUTE", "left"), ("REQ TYPE", "left"), ("TIER", "left"), ("MODEL", "left"), - ("EFFORT", "left"), ("INPUT", "right"), ("COST", "right"), ("VS ASKED", "right")): - table.add_column(name, justify=justify, no_wrap=True) - for when, kind, event, saving in dash.feed: - clock = f"{when:%H:%M:%S}" - if kind == "spend": - route = route_kind(event) - label = f"{ICONS[route]} {route}" + (" (i)" if event.get("background") else "") - cached = event.get("cache_read_tokens") or 0 - tokens = event.get("input_tokens") - size = fmt_tokens(tokens) + (" (c)" if tokens and cached * 2 >= tokens else "") - cost = event.get("cost") - effort = event.get("effort") - table.add_row( - Text(clock, style="dim"), Text(label, style="bold" if route == "invoked" else ""), - type_text(event), event.get("tier") or Text("—", style="dim"), - _model_text(served_model(event, dash.routes, dash.prices)), effort or Text("—", style="dim"), - size, _money(cost) if isinstance(cost, (int, float)) else "—", vs_asked(saving), - ) - elif kind == "deny": - reason = str(event.get("reason", "")).removeprefix("llm-trunk: ") - table.add_row(Text(clock, style="dim"), Text(f"{ICONS['denied']} denied", style="red"), Text(reason[:70], style="red")) - elif kind == "failed": - table.add_row(Text(clock, style="dim"), Text(f"{ICONS['failed']} failed", style="red"), - type_text(event), event.get("tier") or "", Text(f"upstream {event.get('status') or 'error'}", style="red")) - else: - table.add_row(Text(clock, style="dim"), Text(f"{ICONS['expired']} expired", style="yellow"), - Text(f"{event.get('skill_id')} ({event.get('why')}) → back to untagged", style="yellow")) - return Panel(table, title="live", title_align="left") +# (header, right-aligned) +FEED_COLUMNS = (("TIME", False), ("SESSION", False), ("ROUTE", False), ("REQ TYPE", False), ("TIER", False), + ("MODEL", False), ("EFFORT", False), ("INPUT", True), ("COST", True), ("VS ASKED", True)) +GAP = " " + + +def feed_row(dash: Dashboard, when: datetime, kind: str, event: dict, saving: float | None) -> tuple[list, Text | None]: + """(the row's leading cells, the detail that runs on to the end of the line).""" + lead = [Text(f"{when:%H:%M:%S}", style="dim"), _session_text(event.get("session"))] + if kind == "spend": + route = route_kind(event) + label = f"{ICONS[route]} {route}" + (" (i)" if event.get("background") else "") + cached = event.get("cache_read_tokens") or 0 + tokens = event.get("input_tokens") + size = fmt_tokens(tokens) + (" (c)" if tokens and cached * 2 >= tokens else "") + cost = event.get("cost") + effort = event.get("effort") + return [ + *lead, Text(label, style="bold" if route == "invoked" else ""), + type_text(event), event.get("tier") or Text("—", style="dim"), + _model_text(served_model(event, dash.routes, dash.prices)), effort or Text("—", style="dim"), + size, _money(cost) if isinstance(cost, (int, float)) else "—", vs_asked(saving), + ], None + if kind == "deny": + reason = str(event.get("reason", "")).removeprefix("llm-trunk: ") + return [*lead, Text(f"{ICONS['denied']} denied", style="red"), type_text(event)], Text(reason, style="red") + if kind == "failed": + error = " ".join(str(event.get("error") or "").split()) + detail = f"upstream {event.get('status') or 'error'}" + (f": {error}" if error else "") + cells = [*lead, Text(f"{ICONS['failed']} failed", style="red"), type_text(event), event.get("tier") or ""] + return cells, Text(detail, style="red") + detail = f"{event.get('skill_id')} ({event.get('why')}) → back to untagged" + return [*lead, Text(f"{ICONS['expired']} expired", style="yellow"), "", event.get("tier") or ""], Text(detail, style="yellow") + + +def feed_lines(rows: list[tuple[list, Text | None]]) -> list[Text]: + """Lay the rows out in columns sized to what's shown. A row's detail starts + after its last cell and runs on; the panel cuts off whatever doesn't fit.""" + rows = [([cell if isinstance(cell, Text) else Text(cell) for cell in cells], detail) for cells, detail in rows] + widths = [len(name) for name, _ in FEED_COLUMNS] + for cells, _ in rows: + for i, cell in enumerate(cells): + widths[i] = max(widths[i], cell.cell_len) + header = [(Text(name, style="dim"), right) for name, right in FEED_COLUMNS] + lines = [] + for cells, detail in [([cell for cell, _ in header], None), *rows]: + line = Text(no_wrap=True, overflow="ellipsis") + for i, cell in enumerate(cells): + pad = " " * (widths[i] - cell.cell_len) + line.append_text(Text(pad) + cell if FEED_COLUMNS[i][1] else cell + Text(pad)) + if i < len(cells) - 1 or detail: + line.append(GAP) + if detail: + line.append_text(detail) + line.rstrip() + lines.append(line) + return lines + + +def feed_panel(dash: Dashboard, rows: int) -> Panel: + dash.page = max(1, rows) + dash.scroll = min(dash.scroll, dash.max_scroll()) + shown = list(itertools.islice(dash.feed, dash.scroll, dash.scroll + dash.page)) + lines = feed_lines([feed_row(dash, *item) for item in shown]) + total = len(dash.feed) + if dash.scroll: + title = f"live · paused · rows {dash.scroll + 1}–{dash.scroll + len(shown)} of {total}" + if dash.unseen: + title += f" · {dash.unseen} new above" + subtitle = Text("g: back to live", style="bold yellow") + else: + title = f"live · {total} rows" if total > len(shown) else "live" + subtitle = Text("↑↓/wheel: scroll · space/b: page · g/G: newest/oldest · q: quit", style="dim") + return Panel(Group(*lines), title=title, subtitle=subtitle, title_align="left", subtitle_align="right") -def render(dash: Dashboard) -> Layout: +def render(dash: Dashboard, height: int) -> Layout: + # The feed gets whatever height is left: its panel border and header row + # take 3 lines, the rest are request rows. + feed_rows = height - HEADER_HEIGHT - MIDDLE_HEIGHT - 3 layout = Layout() - layout.split_column(Layout(header(dash), size=4), Layout(name="middle", size=12), Layout(feed_panel(dash))) - layout["middle"].split_row(Layout(types_panel(dash)), Layout(sessions_panel(dash))) + layout.split_column( + Layout(header(dash), size=HEADER_HEIGHT), Layout(name="middle", size=MIDDLE_HEIGHT), Layout(feed_panel(dash, feed_rows)) + ) + layout["middle"].split_row(Layout(types_panel(dash), size=TYPES_WIDTH), Layout(sessions_panel(dash))) return layout @@ -326,7 +448,7 @@ def render(dash: Dashboard) -> Layout: def stream(since: str, events: queue.Queue, stop: threading.Event) -> None: - """Follow the gateway log, reconnecting when it restarts; like watch.py.""" + """Follow the gateway log, reconnecting when it restarts without repeating a line.""" resume_after = None while not stop.is_set(): command = ["docker", "compose", "logs", "-f", "-t", "--no-log-prefix", "--since", since, "litellm"] @@ -349,6 +471,40 @@ def stream(since: str, events: queue.Queue, stop: threading.Event) -> None: stop.wait(2) +# Key presses -> dashboard actions. Terminals send the mouse wheel in the +# full-screen view as arrow keys; the letter keys cover a laptop keyboard +# whose Page Up/Home/End the terminal keeps for itself. +KEYS = { + "\x1b[A": "up", "\x1bOA": "up", "k": "up", + "\x1b[B": "down", "\x1bOB": "down", "j": "down", + "\x1b[5~": "pgup", "b": "pgup", + "\x1b[6~": "pgdn", " ": "pgdn", + "\x1b[H": "home", "\x1bOH": "home", "\x1b[1~": "home", "g": "home", + "\x1b[F": "end", "\x1bOF": "end", "\x1b[4~": "end", "G": "end", + "q": "quit", +} + + +def parse_keys(data: str) -> list[str]: + """Raw terminal input -> action names; anything else is skipped.""" + keys, i = [], 0 + while i < len(data): + match = next((seq for seq in sorted(KEYS, key=len, reverse=True) if data.startswith(seq, i)), None) + if match: + keys.append(KEYS[match]) + i += len(match) + else: + i += 1 + return keys + + +def read_keys(fd: int, keys: queue.Queue, stop: threading.Event) -> None: + while not stop.is_set(): + if select.select([fd], [], [], 0.2)[0]: + for key in parse_keys(os.read(fd, 1024).decode(errors="ignore")): + keys.put(key) + + def duration(text: str) -> int: """'30m' / '2h' -> seconds; anything else is an error, not a guess.""" match = re.fullmatch(r"(\d+)([mh])", text.strip()) @@ -370,17 +526,44 @@ def main() -> None: # Without --since, start fresh: earlier sessions would only confuse a new run. since = args.since or datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") threading.Thread(target=stream, args=(since, events, stop), daemon=True).start() + keys: queue.Queue = queue.Queue() + console = Console() + saved_tty = None + if sys.stdin.isatty(): + import termios + import tty + + # Read keys one at a time, unechoed; Ctrl+C still stops the dashboard. + fd = sys.stdin.fileno() + saved_tty = termios.tcgetattr(fd) + tty.setcbreak(fd) + threading.Thread(target=read_keys, args=(fd, keys, stop), daemon=True).start() try: - with Live(render(dash), screen=True, auto_refresh=False, console=Console()) as live: + with Live(render(dash, console.size.height), screen=True, auto_refresh=False, console=console) as live: + # Ask the terminal to send the mouse wheel as arrow keys (most do by default). + console.file.write("\x1b[?1007h") while True: + pressed = [] + try: + pressed.append(keys.get(timeout=0.5)) + while not keys.empty(): + pressed.append(keys.get()) + except queue.Empty: + pass + if "quit" in pressed: + break while not events.empty(): dash.add(*events.get()) - live.update(render(dash), refresh=True) - time.sleep(0.5) + for key in pressed: + dash.press(key) + live.update(render(dash, console.size.height), refresh=True) except KeyboardInterrupt: pass finally: stop.set() + console.file.write("\x1b[?1007l") + if saved_tty is not None: + termios.tcsetattr(sys.stdin.fileno(), termios.TCSADRAIN, saved_tty) if __name__ == "__main__": diff --git a/scripts/events.py b/scripts/events.py new file mode 100644 index 0000000..ea86466 --- /dev/null +++ b/scripts/events.py @@ -0,0 +1,80 @@ +"""The gateway's log format: the `llm-trunk` event lines and how to read them. + +Shared by the dashboard, the report, the rule checker and the scenario runner. +The callback writes one JSON line per outcome (policy/litellm_callback.py, +`_log_event`); tests/test_events.py keeps both sides in step. +""" +import re +import sys +from pathlib import Path + +REPO = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(REPO)) # the other scripts import policy.* after importing this + +STAMP_RE = re.compile(r"^(?P(?P\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2})\S*)\s+(?P.*)$") +# The gateway writes one JSON line per outcome: "llm-trunk : {...}". +EVENT_RE = re.compile(r"llm-trunk (?Pspend|deny|expired|failed): (?P\{.*\})\s*$") + +# Each icon is a single wide (2-cell) code point, so columns stay aligned. +ICONS = { + "invoked": "🔖", "sticky": "📌", "untagged": "⚪", + "unregistered": "🔹", "subagent": "🤖", "compaction": "🧹", "permission": "🔒", "title": "📛", + "denied": "⛔", "failed": "❌", "expired": "⏳", +} + + +def pretty_model(model_id: str) -> str: + # "anthropic/claude-haiku-4-5" -> "Haiku 4.5"; anything else as-is. + name = model_id.split("/")[-1] + # The minor version is 1-2 digits, so a trailing -YYYYMMDD date isn't + # mistaken for one ("claude-opus-4-20250514" -> "Opus 4"). + match = re.fullmatch(r"claude-([a-z]+)-(\d+(?:-\d{1,2})?)(?:-\d{8})?", name) + if not match: + return name + return f"{match.group(1).capitalize()} {match.group(2).replace('-', '.')}" + + +def fmt_tokens(value: int | None) -> str: + if value is None: + return "?" + return f"{value / 1000:.0f}k" if value >= 10_000 else f"{value / 1000:.1f}k" + + +def route_kind(event: dict) -> str: + # Events logged before request types existed have no request_type. + request_type = event.get("request_type") + if request_type == "subagent": + return "subagent" + if request_type == "compaction": + return "compaction" + if event.get("background") == "permission_check": + return "permission" + if event.get("background") == "title": + return "title" + if request_type == "skill" and event.get("unregistered_skill"): + return "unregistered" + if event.get("skill_id") is None: + return "untagged" + return "invoked" if event.get("skill_hash") else "sticky" + + +def request_type_of(event: dict) -> str: + """normal / skill / subagent / compaction / background; events logged + before request types existed are classified from what they do carry.""" + if event.get("request_type"): + return event["request_type"] + if event.get("background"): + return "background" + return "skill" if event.get("skill_hash") else "normal" + + +def tier_text(event: dict) -> str: + """Where the request went: the tier, plus the skill that put it there.""" + tier = event.get("tier") + if event.get("background") == "permission_check": + return f"{tier} · permission" if tier else "passed through" + skill = event.get("unregistered_skill") or event.get("skill_id") + if not tier: + # Old events (per-skill lanes), or a passed-through request. + return skill or "untagged" + return f"{tier} · {skill}" if skill else tier diff --git a/scripts/report.py b/scripts/report.py index dbfa98f..1ddb7ed 100644 --- a/scripts/report.py +++ b/scripts/report.py @@ -6,7 +6,7 @@ python3 scripts/report.py # everything still in the log python3 scripts/report.py --since 24h -Read-only: it reads the same `llm-trunk` log lines as scripts/watch.py and +Read-only: it reads the same `llm-trunk` log lines as the dashboard and totals them per request type, tier, day and session. Costs are LiteLLM's estimates at the configured prices, not an invoice. Docker keeps the log only while the container exists, so a `docker compose up -d` that recreates it starts over. @@ -22,9 +22,9 @@ sys.path.insert(0, str(Path(__file__).resolve().parent)) -from watch import EVENT_RE, REPO, STAMP_RE, request_type_of # noqa: E402 +from events import EVENT_RE, REPO, STAMP_RE, request_type_of # noqa: E402 -from policy.decide import REQUEST_TYPES # noqa: E402 (watch puts the repo on the path) +from policy.decide import REQUEST_TYPES # noqa: E402 (events puts the repo on the path) def parse(lines) -> list[tuple[datetime, str, dict]]: diff --git a/scripts/watch.py b/scripts/watch.py deleted file mode 100755 index c536eeb..0000000 --- a/scripts/watch.py +++ /dev/null @@ -1,307 +0,0 @@ -#!/usr/bin/env python3 -"""Live view of llm-trunk routing decisions, read from the gateway's log. - -Run on the machine hosting the gateway, from anywhere in the repo: - - python3 scripts/watch.py # last 10 minutes, then live - python3 scripts/watch.py --since 1h - -Read-only: it follows `docker compose logs litellm` and renders the -`llm-trunk` lines, reconnecting if the gateway restarts. Set NO_COLOR=1 to -disable colors. -""" -import argparse -import json -import os -import re -import subprocess -import sys -import time -import zlib -from datetime import datetime, timezone -from pathlib import Path - -REPO = Path(__file__).resolve().parent.parent -sys.path.insert(0, str(REPO)) # the other scripts import policy.* after importing this - -STAMP_RE = re.compile(r"^(?P(?P\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2})\S*)\s+(?P.*)$") -# The gateway writes one JSON line per outcome: "llm-trunk : {...}". -EVENT_RE = re.compile(r"llm-trunk (?Pspend|deny|expired|failed): (?P\{.*\})\s*$") - -USE_COLOR = sys.stdout.isatty() and "NO_COLOR" not in os.environ -RED, GREEN, YELLOW, BLUE, MAGENTA, DIM, BOLD, ITALIC = "31", "32", "33", "34", "35", "2", "1", "3" -MODEL_COLORS = {"Opus": MAGENTA, "Sonnet": BLUE, "Haiku": GREEN} -SESSION_COLORS = ["36", "33", "35", "32", "34", "96", "93", "95"] - -# Each icon is a single wide (2-cell) code point, so columns stay aligned. -ICONS = { - "invoked": "🔖", "sticky": "📌", "untagged": "⚪", - "unregistered": "🔹", "subagent": "🤖", "compaction": "🧹", "permission": "🔒", "title": "📛", - "denied": "⛔", "failed": "❌", "expired": "⏳", -} - -# (header, visible width, right-aligned) -COLUMNS = [ - ("TIME", 8, False), - ("SESSION", 8, False), - ("ROUTE", 15, False), - ("MODEL · EFFORT", 18, False), - ("TIER · SKILL", 22, False), - ("STICKY", 8, True), - ("INPUT", 9, True), - ("COST", 7, True), -] -GAP = " " - - -def color(code: str | None, text: str) -> str: - return f"\033[{code}m{text}\033[0m" if USE_COLOR and code else text - - -def cell(index: int, text: str, code: str | None = None, *, wide_chars: int = 0) -> str: - # Pad to the column's visible width (a wide icon takes 2 cells, 1 char). - _, width, right = COLUMNS[index] - width -= wide_chars - return color(code, text.rjust(width) if right else text.ljust(width)) - - -def pretty_model(model_id: str) -> str: - # "anthropic/claude-haiku-4-5" -> "Haiku 4.5"; anything else as-is. - name = model_id.split("/")[-1] - # The minor version is 1-2 digits, so a trailing -YYYYMMDD date isn't - # mistaken for one ("claude-opus-4-20250514" -> "Opus 4"). - match = re.fullmatch(r"claude-([a-z]+)-(\d+(?:-\d{1,2})?)(?:-\d{8})?", name) - if not match: - return name - return f"{match.group(1).capitalize()} {match.group(2).replace('-', '.')}" - - -def load_models() -> dict[str, str]: - # Alias -> display name, from litellm/config.yaml (stdlib-only parse). - models, alias = {}, None - for line in (REPO / "litellm" / "config.yaml").read_text().splitlines(): - if match := re.match(r"\s*- model_name:\s*(\S+)", line): - alias = match.group(1) - elif (match := re.match(r"\s+model:\s*(\S+)", line)) and alias: - models[alias] = pretty_model(match.group(1)) - alias = None - return models - - -def local_time(second: str) -> str: - utc = datetime.strptime(second, "%Y-%m-%dT%H:%M:%S").replace(tzinfo=timezone.utc) - return utc.astimezone().strftime("%H:%M:%S") - - -def session_cell(session: str | None) -> str: - if not session: - return cell(1, "—", DIM) - return cell(1, session, SESSION_COLORS[zlib.crc32(session.encode()) % len(SESSION_COLORS)]) - - -def route_cell(kind: str, code: str | None = None, *, background: bool = False) -> str: - # "(i)" marks Claude Code's own background calls (session title, away - # summary): they never refresh a sticky timer. - label = f"{ICONS[kind]} {kind}" - if not background: - return cell(2, label, code, wide_chars=1) - _, width, _ = COLUMNS[2] - visible = len(label) + 1 + len(" (i)") # the icon is 2 cells wide - return color(code, label) + " " + color(DIM + ";" + ITALIC, "(i)") + " " * max(0, width - visible) - - -def fmt_tokens(value: int | None) -> str: - if value is None: - return "?" - return f"{value / 1000:.0f}k" if value >= 10_000 else f"{value / 1000:.1f}k" - - -def input_cell(tokens: int | None, cached: int | None) -> str: - # "(c)": most of the input was read from Anthropic's prompt cache (billed - # at ~10%). In practice a request is either ~0% or ~90-100% cached. - _, width, _ = COLUMNS[6] - value = fmt_tokens(tokens).rjust(width - 4) - if tokens and cached and cached * 2 >= tokens: - return value + " " + color(DIM, "(c)") - return value + " " - - -def fmt_left(seconds: int | None) -> str: - if seconds is None: - return "—" - return f"{(seconds + 59) // 60}m left" if seconds >= 60 else f"{seconds}s left" - - -def route_kind(event: dict) -> str: - # Events logged before request types existed have no request_type. - request_type = event.get("request_type") - if request_type == "subagent": - return "subagent" - if request_type == "compaction": - return "compaction" - if event.get("background") == "permission_check": - return "permission" - if event.get("background") == "title": - return "title" - if request_type == "skill" and event.get("unregistered_skill"): - return "unregistered" - if event.get("skill_id") is None: - return "untagged" - return "invoked" if event.get("skill_hash") else "sticky" - - -def request_type_of(event: dict) -> str: - """normal / skill / subagent / compaction / background; events logged - before request types existed are classified from what they do carry.""" - if event.get("request_type"): - return event["request_type"] - if event.get("background"): - return "background" - return "skill" if event.get("skill_hash") else "normal" - - -def tier_text(event: dict) -> str: - """Where the request went: the tier, plus the skill that put it there.""" - tier = event.get("tier") - if event.get("background") == "permission_check": - return f"{tier} · permission" if tier else "passed through" - skill = event.get("unregistered_skill") or event.get("skill_id") - if not tier: - # Old events (per-skill lanes), or a passed-through request. - return skill or "untagged" - return f"{tier} · {skill}" if skill else tier - - -def model_cells(event: dict, models: dict[str, str]) -> list[str]: - # The model that answered, as logged; the tier's current model only for - # old lines that didn't log one. - alias = event.get("alias") or "?" - model = pretty_model(event["model"]) if event.get("model") else models.get(alias) or pretty_model(alias) - text = f"{model} · {event['effort']}" if event.get("effort") else model - _, width, _ = COLUMNS[4] - where = tier_text(event) - return [ - cell(3, text, MODEL_COLORS.get(model.split(" ")[0])), - cell(4, where if len(where) <= width else where[: width - 1] + "…"), - ] - - -def render_spend(clock: str, event: dict, models: dict[str, str]) -> str: - kind = route_kind(event) - cost = event.get("cost") - return GAP.join( - [ - cell(0, clock, DIM), - session_cell(event.get("session")), - route_cell(kind, BOLD if kind == "invoked" else None, background=bool(event.get("background"))), - *model_cells(event, models), - cell(5, fmt_left(event.get("sticky_left_s")), DIM if kind == "untagged" else YELLOW), - input_cell(event.get("input_tokens"), event.get("cache_read_tokens")), - cell(7, f"${cost:.3f}" if isinstance(cost, (int, float)) else "—"), - ] - ) - - -def render(message: str, clock: str, models: dict[str, str]) -> str | None: - match = EVENT_RE.search(message) - if not match: - return None - try: - event = json.loads(match.group("json")) - except ValueError: - return None - kind = match.group("kind") - background = bool(event.get("background")) - lead = [cell(0, clock, DIM), session_cell(event.get("session"))] - if kind == "spend": - return render_spend(clock, event, models) - if kind == "deny": - reason = str(event.get("reason", "")).removeprefix("llm-trunk: ") - return GAP.join([*lead, route_cell("denied", RED, background=background), color(RED, reason)]) - if kind == "failed": - status = event.get("status") - error = f"{status}: " if status else "" - error += " ".join(str(event.get("error") or "upstream error").split())[:120] - return GAP.join( - [*lead, route_cell("failed", RED, background=background), *model_cells(event, models), color(RED, error)] - ) - detail = f"{event.get('skill_id')} ({event.get('why')}) → back to untagged" - return GAP.join([*lead, route_cell("expired", YELLOW), color(YELLOW, detail)]) - - -def header() -> str: - legend = " ".join(f"{icon} {kind}" for kind, icon in ICONS.items()) + " (i) background call (c) cached input" - columns = GAP.join( - name.rjust(width) if right else name.ljust(width) for name, width, right in COLUMNS - ) - return "\n".join( - [ - color(BOLD, "🔀 llm-trunk · live routing") + color(DIM, " Ctrl+C to stop"), - color(DIM, legend), - "", - color(DIM, columns), - color(DIM, "─" * len(columns)), - ] - ) - - -def follow(since: str, resume_after: str | None, models: dict[str, str]) -> tuple[str | None, str | None]: - """Stream the log; return the last stamp seen and docker's last error line.""" - command = ["docker", "compose", "logs", "-f", "-t", "--no-log-prefix", "--since", since, "litellm"] - process = subprocess.Popen( - command, cwd=REPO, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True, bufsize=1 - ) - assert process.stdout is not None - last_error = None - try: - for line in process.stdout: - match = STAMP_RE.match(line.rstrip("\n")) - if not match: - # Log lines are always timestamped (-t); anything else is - # docker itself talking -- e.g. an error. - last_error = line.strip() or last_error - continue - if resume_after and match.group("stamp") <= resume_after: - continue - resume_after = match.group("stamp") - rendered = render(match.group("message"), local_time(match.group("second")), models) - if rendered: - print(rendered, flush=True) - finally: - process.terminate() - process.wait() - return resume_after, last_error - - -def main() -> None: - parser = argparse.ArgumentParser(description="Live view of llm-trunk routing decisions.") - parser.add_argument("--since", default="10m", help="show history this far back (default: 10m)") - args = parser.parse_args() - - models = load_models() - print(header(), flush=True) - since, resume_after, notice = args.since, None, None - try: - while True: - # The stream ends when the gateway restarts (or isn't up yet): - # resume after the last line shown, so nothing repeats or is lost. - last, error = follow(since, resume_after, models) - if last and last != resume_after: - since = resume_after = last - message = "gateway log ended — reconnecting" - elif error: - message = f"docker: {error}" - else: - message = "waiting for the gateway (docker compose up -d)" - if message != notice: - print(color(DIM, f" … {message}"), flush=True) - notice = message - time.sleep(2) - except KeyboardInterrupt: - print() - except FileNotFoundError: - sys.exit("error: docker not found — run this on the machine hosting the gateway") - - -if __name__ == "__main__": - main() diff --git a/tests/sanity/manual.py b/tests/sanity/manual.py index df4ec4d..52d2b89 100644 --- a/tests/sanity/manual.py +++ b/tests/sanity/manual.py @@ -7,16 +7,16 @@ Start a fresh Claude Code session for tests 1–6 and run them in order. -**Reading the dashboard:** in the live table, the route icon shows what kind of request it was. `(i)` marks a request Claude Code made on its own, not something you typed. The next column is "tier · skill", and the model column shows the model and effort level. +**Reading the dashboard:** in the live feed, the route icon shows what kind of request it was. `(i)` marks a request Claude Code made on its own, not something you typed. REQ TYPE names the skill when the request invokes one, followed by the TIER, MODEL and EFFORT columns. The sessions pane shows each session's prompt-cache countdown (CACHE) and its sticky-tier countdown (STICKY). | # | Do this | Pass criteria (dashboard) | Why | |---|---|---|---| | 1 | Send a plain question, e.g. "What does a load balancer do?" | A `📛 title (i)` row on **light** with no effort shown, plus `⚪ untagged` on **light · Haiku 4.5 · low** | Plain chat is the "untagged frame": it goes to the lowest tier. The title is Claude Code naming the session; it's ~1k tokens and needs no thinking. | -| 2 | `/design-review Move sessions to Redis` | `🔖 invoked` · **complex · design-review** · Opus 5.5 · high · **10m left** | The skill's SHA-256 hash matches its entry in `catalog.yaml`, and that entry assigns the complex tier. This starts the session's 10-minute "sticky" timer. | -| 3 | Within 10 min, ask a follow-up ("Which risk first?") | `📌 sticky` on **complex · design-review**, timer back to **10m**. Any `(i)` suggestion rows are also on complex, but they don't reset the timer. | Follow-ups keep the skill's tier (bounded stickiness). Only your own turns refresh the timer. Background calls stay on the session tier so the prompt cache isn't rebuilt on another model. | +| 2 | `/design-review Move sessions to Redis` | `🔖 invoked` · `skill (design-review)` · **complex** · Opus 5.5 · high. The sessions pane's STICKY shows **📌 10:00 design-review**, counting down | The skill's SHA-256 hash matches its entry in `catalog.yaml`, and that entry assigns the complex tier. This starts the session's 10-minute "sticky" timer. | +| 3 | Within 10 min, ask a follow-up ("Which risk first?") | `📌 sticky` on **complex**, STICKY back to **10:00**. Any `(i)` suggestion rows are also on complex, but they don't reset the timer. | Follow-ups keep the skill's tier (bounded stickiness). Only your own turns refresh the timer. Background calls stay on the session tier so the prompt cache isn't rebuilt on another model. | | 4 | Still in that session: "Use a subagent to list the folders in .claude/skills" | `🤖 subagent` rows on **moderate · Sonnet 5**, while the main session stays complex | Subagents run one tier below the session, and each keeps the tier it started on. | -| 5 | `/commit-message Fixed a typo`, then a follow-up | `🔖 invoked` on **light · commit-message**, then `📌 sticky` on light | A newly invoked registered skill sets the session's tier, even if that's lower than before. Stickiness follows the most recent skill. | -| 6 | `/personal-notes check redis ttl`, then a follow-up | `🔹 unregistered` on **light · personal-notes**, then `⚪ untagged` on light with no timer | A skill not in the catalog gets no tier of its own and ends the sticky route. Unknown code never gets an expensive model. | +| 5 | `/commit-message Fixed a typo`, then a follow-up | `🔖 invoked` · `skill (commit-message)` on **light**, then `📌 sticky` on light | A newly invoked registered skill sets the session's tier, even if that's lower than before. Stickiness follows the most recent skill. | +| 6 | `/personal-notes check redis ttl`, then a follow-up | `🔹 unregistered` · `skill (personal-notes)` on **light**, then `⚪ untagged` on light, and STICKY shows `—` | A skill not in the catalog gets no tier of its own and ends the sticky route. Unknown code never gets an expensive model. | | 7 | In `company-client/.claude/skills/change-review/SKILL.md`, add one line. Run `/change-review x`. Then undo the edit and run it again. | First run: `⛔ denied`, and "changed since it was hashed" appears in both the dashboard and Claude Code (`API Error: 400 …`). After the undo: `🔖 invoked` on moderate. | Any edit changes the hash, so an edited skill can't keep its tier until someone re-hashes it. The request is refused before it reaches Anthropic, so it costs $0. | | 8 | In a light session, paste a file of more than about 250 KB (over 64k tokens), then run `/compact` | The paste shows `⛔ denied … exceeds the light tier cap (64000)`, and Claude Code shows the same message. `/compact` then shows `🧹 compaction` and succeeds. | Each tier has a maximum input size (`max_input`), so a huge paste can't run up a bill. Compaction is exempt, so a session over the limit can always be shrunk. | | 9 | Switch to auto mode (Shift+Tab) and ask for something that edits a file | `🔒 permission (i)` rows on the tier running the model Claude Code asked for (with the default Opus 5.5: **complex**). A skill session's timer doesn't change. | The permission check is a safety check, so it's never moved to a weaker model. With a key limited by `allowed_skills` it stays within that key's tiers (the ceiling fix). Background calls never extend stickiness. | diff --git a/tests/test_dashboard.py b/tests/test_dashboard.py index b9f441a..7640cd7 100644 --- a/tests/test_dashboard.py +++ b/tests/test_dashboard.py @@ -109,9 +109,9 @@ def test_cache_clock_warm_then_cold(): assert dash.cache_state(dash.sessions["s1"])[0] is False -def _screen(dash) -> str: - console = Console(record=True, width=120, height=40, color_system=None) - console.print(dashboard.render(dash)) +def _screen(dash, width=120) -> str: + console = Console(record=True, width=width, height=40, color_system=None) + console.print(dashboard.render(dash, 40)) return console.export_text() @@ -189,7 +189,7 @@ def test_warm_session_line_is_not_truncated(): dash = _dashboard() dash.add(T0, "spend", _spend(alias="complex", input_tokens=53_000)) screen = _screen(dash) - assert "next $0.011 · cold $0.265" in screen + assert "NEXT MESSAGE" in screen and "$0.011 · cold $0.265" in screen def test_spend_pane_groups_by_request_type(): @@ -250,3 +250,119 @@ def test_window_durations(text, seconds): def test_window_rejects_what_it_would_misread(text): with pytest.raises(argparse.ArgumentTypeError, match="use minutes or hours"): dashboard.duration(text) + + +def test_feed_names_each_rows_session(): + dash = _dashboard() + dash.add(T0, "spend", _spend(session="a1b2c3d4")) + dash.add(T0, "deny", {"reason": "llm-trunk: too big", "session": "ffee0011", "request_type": "normal"}) + feed = _screen(dash).split("─ live")[1] + assert "SESSION" in feed and "a1b2c3d4" in feed and "ffee0011" in feed + + +def test_deny_reason_runs_to_the_end_of_the_line(): + dash = _dashboard() + reason = ("llm-trunk: skill 'change-review' changed since it was hashed (got 3f2a9c41d0e7, catalog has " + "8b1e77c2a9d4) — re-run scripts/hash_skill.py and update catalog.yaml") + dash.add(T0, "spend", _spend()) + dash.add(T0, "deny", {"code": "stale_hash", "reason": reason, "session": "s1", "request_type": "skill"}) + assert reason.removeprefix("llm-trunk: ") in _screen(dash, width=200) + narrow = _screen(dash) # cut off at the edge, without squeezing the other rows + assert "changed since it was hashed" in narrow and "update catalog.yaml" not in narrow + assert "40k (c) $0.007" in narrow + + +def _busy(clock, rows): + dash = _dashboard(clock) + for i in range(rows): + dash.add(T0, "spend", _spend(session=f"r{i:04d}")) + return dash + + +def test_feed_fills_the_screen_and_keeps_history(): + dash = _busy(Clock(), 200) + screen = _screen(dash) # 40 lines: 21 request rows + assert "r0199" in screen and "r0179" in screen and "r0178" not in screen + assert "live · 200 rows" in screen + assert len(dash.feed) == 200 + + +def test_scrolling_back_through_the_feed(): + dash = _busy(Clock(), 200) + _screen(dash) + dash.press("pgdn") + screen = _screen(dash) + assert "r0178" in screen and "r0179" not in screen + assert "paused · rows 22–42 of 200" in screen and "g: back to live" in screen + dash.press("end") + assert "r0000" in _screen(dash) + dash.press("down") # already at the oldest row + assert dash.scroll == 200 - dash.page + dash.press("home") + assert dash.scroll == 0 and "r0199" in _screen(dash) + dash.press("up") # already at the newest row + assert dash.scroll == 0 + + +def test_new_rows_do_not_move_a_scrolled_view(): + clock = Clock() + dash = _busy(clock, 100) + _screen(dash) + dash.press("down") + dash.press("down") + dash.add(T0, "spend", _spend(session="new00001")) + screen = _screen(dash) + assert "r0097" in screen.split("\n")[18] # still the first row shown + assert "new00001" not in screen and "1 new above" in screen + dash.press("home") + screen = _screen(dash) + assert "new00001" in screen and "new above" not in screen + + +def test_scrolling_is_harmless_on_a_short_feed(): + dash = _busy(Clock(), 3) + _screen(dash) + for key in ("down", "pgdn", "end"): + dash.press(key) + assert dash.scroll == 0 + + +@pytest.mark.parametrize( + ("data", "keys"), + [ + ("\x1b[A\x1b[A\x1b[B", ["up", "up", "down"]), # arrows, or the mouse wheel + ("\x1bOA\x1bOB", ["up", "down"]), # arrows in application mode + ("\x1b[5~\x1b[6~ b", ["pgup", "pgdn", "pgdn", "pgup"]), + ("gG\x1b[H\x1b[F", ["home", "end", "home", "end"]), + ("jkxq", ["down", "up", "quit"]), + ("\x1b[Z", []), # anything else is ignored + ], +) +def test_keys(data, keys): + assert dashboard.parse_keys(data) == keys + + +def test_sessions_pane_shows_the_sticky_timer(): + clock = Clock() + dash = _dashboard(clock) + dash.add(T0, "spend", _spend(request_type="skill", skill_id="design-review", skill_hash="h", tier="complex", + alias="complex", sticky_left_s=600)) + clock.now += 70 + assert "📌 8:50 design-review" in _screen(dash, width=160) + assert "📌 8:50" in _screen(dash) # a narrow terminal shortens the skill name, never the timer + # A background call reports the timer without refreshing it; a subagent reports none. + dash.add(T0, "spend", _spend(request_type="background", background="title", skill_id="design-review", sticky_left_s=600)) + dash.add(T0, "spend", _spend(request_type="subagent", tier="moderate")) + assert dash.sticky_left("s1") == ("design-review", 530) + clock.now += 600 + assert dash.sticky_left("s1") is None + + +def test_sticky_timer_ends_with_the_route(): + dash = _dashboard() + dash.add(T0, "spend", _spend(request_type="skill", skill_id="plan", skill_hash="h", tier="complex", sticky_left_s=600)) + dash.add(T0, "expired", {"skill_id": "plan", "tier": "complex", "why": "idle", "session": "s1"}) + assert dash.sticky_left("s1") is None + dash.add(T0, "spend", _spend(request_type="skill", skill_id="plan", skill_hash="h", tier="complex", sticky_left_s=600)) + dash.add(T0, "spend", _spend(request_type="skill", unregistered_skill="notes", sticky_left_s=None)) + assert dash.sticky_left("s1") is None diff --git a/tests/test_events.py b/tests/test_events.py index 0b731a5..a9296fc 100644 --- a/tests/test_events.py +++ b/tests/test_events.py @@ -1,8 +1,8 @@ -"""The log lines the callback writes are the contract with scripts/watch.py: -each outcome must still be picked up and rendered by the watcher.""" +"""The log lines the callback writes are the contract with scripts/events.py: +each outcome must still be parsed, and shown on the dashboard.""" import asyncio -import importlib.util import json +import sys import types import pytest @@ -18,12 +18,25 @@ unregistered_turn, user, ) +from rich.console import Console -spec = importlib.util.spec_from_file_location("watch", REPO / "scripts" / "watch.py") -watch = importlib.util.module_from_spec(spec) -spec.loader.exec_module(watch) +sys.path.insert(0, str(REPO / "scripts")) -MODELS = {"complex": "Opus 5.5", "moderate": "Sonnet 5", "light": "Haiku 4.5"} +import dashboard # noqa: E402 +import events # noqa: E402 +from report import parse # noqa: E402 + +ROUTES = {"complex": "anthropic/claude-opus-5-5", "moderate": "anthropic/claude-sonnet-5", "light": "anthropic/claude-haiku-4-5"} + + +def render(*lines: str) -> str: + """The dashboard's screen after these log lines, as docker prints them.""" + dash = dashboard.Dashboard(dashboard.load_prices(), ROUTES, 300) + for item in parse(f"2026-09-26T11:00:00.000000000Z {line}" for line in lines): + dash.add(*item) + console = Console(record=True, width=200, height=40, color_system=None) + console.print(dashboard.render(dash, 40)) + return console.export_text().split("─ live")[1] # the feed def _log_lines(output: str) -> list[str]: @@ -31,8 +44,8 @@ def _log_lines(output: str) -> list[str]: def _event(line: str) -> tuple[str, dict]: - match = watch.EVENT_RE.search(line) - assert match, f"watcher wouldn't recognize: {line}" + match = events.EVENT_RE.search(line) + assert match, f"events.py wouldn't recognize: {line}" return match.group("kind"), json.loads(match.group("json")) @@ -64,41 +77,41 @@ def test_spend_event_is_rendered(gateway, capsys): assert event["model"] == "claude-opus-5-5-20260915" # what actually answered assert event["requested_model"] == "claude-sonnet-5" # what the client asked for assert event["session_tier"] == "light" and event["cost"] == 0.0123 - rendered = watch.render(line, "12:00:00", MODELS) - assert "invoked" in rendered and "Opus 5.5 · high" in rendered and "complex · plan" in rendered - assert "(c)" in rendered and "$0.012" in rendered + rendered = render(line) + assert "🔖 invoked" in rendered and "skill (plan)" in rendered and "complex" in rendered + assert "Opus 5.5" in rendered and "high" in rendered and "41k (c)" in rendered and "$0.012" in rendered @pytest.mark.parametrize( - ("build", "route", "where"), + ("build", "route", "request_type", "where"), [ - (lambda: request(skill_turn("plan"), assistant(), user("Next?")), "sticky", "complex · plan"), - (lambda: subagent_request(), "subagent", "moderate"), - (lambda: compaction_request(skill_turn("plan"), assistant(), user("go")), "compaction", "complex · plan"), - (lambda: request(skill_turn("plan"), assistant(), unregistered_turn()), "unregistered", "light · personal-notes"), - (lambda: permission_check_request(), "permission", "moderate · permission"), - (lambda: request(user("fix login"), tools=False), "title", "light"), + (lambda: request(skill_turn("plan"), assistant(), user("Next?")), "sticky", "normal", "complex · plan"), + (lambda: subagent_request(), "subagent", "subagent", "moderate"), + (lambda: compaction_request(skill_turn("plan"), assistant(), user("go")), "compaction", "compaction", "complex · plan"), + (lambda: request(skill_turn("plan"), assistant(), unregistered_turn()), "unregistered", "skill (personal-notes)", + "light · personal-notes"), + (lambda: permission_check_request(), "permission (i)", "internal", "moderate · permission"), + (lambda: request(user("fix login"), tools=False), "title (i)", "internal", "light"), ], ) -def test_every_request_type_is_rendered(gateway, capsys, build, route, where): +def test_every_request_type_is_rendered(gateway, capsys, build, route, request_type, where): gateway.send(request(skill_turn("plan"))) # puts the session on the complex tier - rendered = watch.render(_spend_line(gateway, capsys, gateway.send(build())), "12:00:00", MODELS) - assert f"{watch.ICONS[route]} {route}" in rendered - assert where in rendered + line = _spend_line(gateway, capsys, gateway.send(build())) + _, event = _event(line) + rendered = render(line) + assert f"{events.ICONS[route.split()[0]]} {route}" in rendered + assert request_type in rendered and event["tier"] in rendered + assert where in events.tier_text(event) # the scenario runner's "tier · skill" def test_permission_check_shows_the_model_claude_code_chose(gateway, capsys): - rendered = watch.render(_spend_line(gateway, capsys, gateway.send(permission_check_request())), "12:00:00", MODELS) - assert "Sonnet 5" in rendered + assert "Sonnet 5" in render(_spend_line(gateway, capsys, gateway.send(permission_check_request()))) -def test_long_tier_text_is_truncated_to_the_column(gateway, capsys): +def test_long_skill_names_are_shown_in_full(): event = {"request_type": "skill", "tier": "complex", "skill_id": "incident-postmortem", "skill_hash": "h", "alias": "complex"} - text = watch.tier_text(event) - assert text == "complex · incident-postmortem" - rendered = watch.render("llm-trunk spend: " + json.dumps(event), "12:00:00", MODELS) - _, width, _ = watch.COLUMNS[4] - assert text[: width - 1] + "…" in rendered + assert events.tier_text(event) == "complex · incident-postmortem" + assert "skill (incident-postmortem)" in render("llm-trunk spend: " + json.dumps(event)) def test_deny_event_is_rendered(gateway, capsys): @@ -108,7 +121,7 @@ def test_deny_event_is_rendered(gateway, capsys): (line,) = _log_lines(capsys.readouterr().out) kind, event = _event(line) assert (kind, event["code"], event["request_type"]) == ("deny", "input_cap", "normal") - assert "light tier cap" in watch.render(line, "12:00:00", MODELS) + assert "light tier cap" in render(line) def test_failed_event_is_rendered(gateway, capsys): @@ -119,7 +132,7 @@ def test_failed_event_is_rendered(gateway, capsys): (line,) = _log_lines(capsys.readouterr().out) kind, event = _event(line) assert (kind, event["status"], event["tier"]) == ("failed", 529, "light") - assert "529" in watch.render(line, "12:00:00", MODELS) + assert "upstream 529" in render(line) def test_expired_event_is_rendered(gateway, clock, capsys): @@ -130,7 +143,7 @@ def test_expired_event_is_rendered(gateway, clock, capsys): (line,) = _log_lines(capsys.readouterr().out) kind, event = _event(line) assert (kind, event["skill_id"], event["tier"]) == ("expired", "plan", "complex") - assert "back to untagged" in watch.render(line, "12:00:00", MODELS) + assert "plan (idle) → back to untagged" in render(line) def test_old_event_lines_still_render(): @@ -138,12 +151,13 @@ def test_old_event_lines_still_render(): old = {"skill_id": "qa-bug-logging", "skill_hash": None, "alias": "qa-bug-logging", "effort": "low", "session": "05887d29", "sticky_left_s": 300, "background": None, "input_tokens": 53000, "cache_read_tokens": 52000, "cost": 0.012} - rendered = watch.render("llm-trunk spend: " + json.dumps(old), "11:34:16", {"qa-bug-logging": "Sonnet 5"}) - assert "sticky" in rendered and "qa-bug-logging" in rendered and "Sonnet 5 · low" in rendered + rendered = render("llm-trunk spend: " + json.dumps(old)) + assert "📌 sticky" in rendered and "05887d29" in rendered and "low" in rendered and "53k (c)" in rendered + assert events.tier_text(old) == "qa-bug-logging" -def test_watcher_ignores_unknown_fields(gateway, capsys): +def test_unknown_fields_are_ignored(gateway, capsys): line = _spend_line(gateway, capsys, gateway.send(request(skill_turn("plan")))) kind, event = _event(line) extended = f"llm-trunk {kind}: " + json.dumps({**event, "some_future_field": 1}) - assert watch.render(extended, "12:00:00", MODELS) == watch.render(line, "12:00:00", MODELS) + assert render(extended) == render(line) diff --git a/tests/test_report.py b/tests/test_report.py index 16a4930..9d2d8d4 100644 --- a/tests/test_report.py +++ b/tests/test_report.py @@ -1,4 +1,4 @@ -"""scripts/report.py: totals from the same log lines the watcher reads.""" +"""scripts/report.py: totals from the same log lines the dashboard reads.""" import json import sys import time