Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 1 addition & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -210,8 +210,7 @@ Run these in the `llm-trunk` folder on the gateway's machine.

| Command | What it does |
|---|---|
| `python3 scripts/dashboard.py` | Live dashboard: savings, spend per request type, prompt-cache countdown per session, live feed. Starts empty; `--since 1h` adds history |
| `python3 scripts/watch.py` | Live log, one line per request |
| `python3 scripts/dashboard.py` | Live dashboard: savings, spend per request type, prompt-cache and sticky-tier countdowns per session, and a feed with one row per request (scroll with ↑/↓ or the mouse wheel, `g` for newest). Starts empty; `--since 1h` adds history |
| `python3 scripts/report.py --since 24h` | Cost totals per request type, tier, day and session |
| `python3 scripts/check_rules.py --since 1h` | Checks every logged request against the routing rules; exits with 1 on a violation |
| `python3 scenarios/run.py --quick` | Drives a scripted Claude Code session through the gateway and checks each step's tier and answer (about $1–2 per run) |
Expand Down
4 changes: 2 additions & 2 deletions policy/litellm_callback.py
Original file line number Diff line number Diff line change
Expand Up @@ -208,7 +208,7 @@ def _latest_user_blocks(data: dict) -> list[str]:

def _log_event(kind: str, fields: dict) -> None:
# One JSON line per outcome -- spend, deny, expired, failed -- which
# scripts/watch.py renders. Structured so a client-supplied value (e.g. a
# scripts/events.py parses. Structured so a client-supplied value (e.g. a
# conversation id with spaces) can't break or hide a line.
_log(f"llm-trunk {kind}: " + json.dumps(fields))

Expand Down Expand Up @@ -755,7 +755,7 @@ async def async_log_success_event(self, kwargs, response_obj, start_time, end_ti
"input_tokens": _usage_value(usage, "prompt_tokens", "input_tokens"),
"output_tokens": _usage_value(usage, "completion_tokens", "output_tokens"),
# Reads only: a cache *write* is billed at 1.25x, the opposite
# of what the watcher's "(c)" tag claims.
# of what the dashboard's "(c)" tag claims.
"cache_read_tokens": _usage_value(usage, "cache_read_input_tokens"),
# Both are included in input_tokens (LiteLLM adds them in).
"cache_write_tokens": _usage_value(usage, "cache_creation_input_tokens"),
Expand Down
2 changes: 1 addition & 1 deletion scenarios/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -46,7 +46,7 @@
import check_rules # noqa: E402
import dashboard # noqa: E402
from report import parse # noqa: E402
from watch import pretty_model, tier_text # noqa: E402
from events import pretty_model, tier_text # noqa: E402

GATEWAY = "http://127.0.0.1:4000"
QUIET_SECONDS = 8 # a suggestion lands a few seconds after the reply
Expand Down
2 changes: 1 addition & 1 deletion scripts/check_rules.py
Original file line number Diff line number Diff line change
Expand Up @@ -36,7 +36,7 @@
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))

from report import parse # noqa: E402
from watch import REPO # noqa: E402
from events import REPO # noqa: E402

from policy.decide import lowest, one_down # noqa: E402
from policy.models import model_key # noqa: E402
Expand Down
291 changes: 237 additions & 54 deletions scripts/dashboard.py

Large diffs are not rendered by default.

80 changes: 80 additions & 0 deletions scripts/events.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,80 @@
"""The gateway's log format: the `llm-trunk` event lines and how to read them.

Shared by the dashboard, the report, the rule checker and the scenario runner.
The callback writes one JSON line per outcome (policy/litellm_callback.py,
`_log_event`); tests/test_events.py keeps both sides in step.
"""
import re
import sys
from pathlib import Path

REPO = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(REPO)) # the other scripts import policy.* after importing this

STAMP_RE = re.compile(r"^(?P<stamp>(?P<second>\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2})\S*)\s+(?P<message>.*)$")
# The gateway writes one JSON line per outcome: "llm-trunk <kind>: {...}".
EVENT_RE = re.compile(r"llm-trunk (?P<kind>spend|deny|expired|failed): (?P<json>\{.*\})\s*$")

# Each icon is a single wide (2-cell) code point, so columns stay aligned.
ICONS = {
"invoked": "🔖", "sticky": "📌", "untagged": "⚪",
"unregistered": "🔹", "subagent": "🤖", "compaction": "🧹", "permission": "🔒", "title": "📛",
"denied": "⛔", "failed": "❌", "expired": "⏳",
}


def pretty_model(model_id: str) -> str:
# "anthropic/claude-haiku-4-5" -> "Haiku 4.5"; anything else as-is.
name = model_id.split("/")[-1]
# The minor version is 1-2 digits, so a trailing -YYYYMMDD date isn't
# mistaken for one ("claude-opus-4-20250514" -> "Opus 4").
match = re.fullmatch(r"claude-([a-z]+)-(\d+(?:-\d{1,2})?)(?:-\d{8})?", name)
if not match:
return name
return f"{match.group(1).capitalize()} {match.group(2).replace('-', '.')}"


def fmt_tokens(value: int | None) -> str:
if value is None:
return "?"
return f"{value / 1000:.0f}k" if value >= 10_000 else f"{value / 1000:.1f}k"


def route_kind(event: dict) -> str:
# Events logged before request types existed have no request_type.
request_type = event.get("request_type")
if request_type == "subagent":
return "subagent"
if request_type == "compaction":
return "compaction"
if event.get("background") == "permission_check":
return "permission"
if event.get("background") == "title":
return "title"
if request_type == "skill" and event.get("unregistered_skill"):
return "unregistered"
if event.get("skill_id") is None:
return "untagged"
return "invoked" if event.get("skill_hash") else "sticky"


def request_type_of(event: dict) -> str:
"""normal / skill / subagent / compaction / background; events logged
before request types existed are classified from what they do carry."""
if event.get("request_type"):
return event["request_type"]
if event.get("background"):
return "background"
return "skill" if event.get("skill_hash") else "normal"


def tier_text(event: dict) -> str:
"""Where the request went: the tier, plus the skill that put it there."""
tier = event.get("tier")
if event.get("background") == "permission_check":
return f"{tier} · permission" if tier else "passed through"
skill = event.get("unregistered_skill") or event.get("skill_id")
if not tier:
# Old events (per-skill lanes), or a passed-through request.
return skill or "untagged"
return f"{tier} · {skill}" if skill else tier
6 changes: 3 additions & 3 deletions scripts/report.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@
python3 scripts/report.py # everything still in the log
python3 scripts/report.py --since 24h

Read-only: it reads the same `llm-trunk` log lines as scripts/watch.py and
Read-only: it reads the same `llm-trunk` log lines as the dashboard and
totals them per request type, tier, day and session. Costs are LiteLLM's estimates
at the configured prices, not an invoice. Docker keeps the log only while the
container exists, so a `docker compose up -d` that recreates it starts over.
Expand All @@ -22,9 +22,9 @@

sys.path.insert(0, str(Path(__file__).resolve().parent))

from watch import EVENT_RE, REPO, STAMP_RE, request_type_of # noqa: E402
from events import EVENT_RE, REPO, STAMP_RE, request_type_of # noqa: E402

from policy.decide import REQUEST_TYPES # noqa: E402 (watch puts the repo on the path)
from policy.decide import REQUEST_TYPES # noqa: E402 (events puts the repo on the path)


def parse(lines) -> list[tuple[datetime, str, dict]]:
Expand Down
Loading
Loading