feat(metrics): lokaler LLM-Verbrauch + Claude-Ersparnis in Diagnostic

metrics.jsonl-Eintraege tragen jetzt 'source' (claude|local|fast-path).
- log_local_call(): echte usage-Tokens vom Adapter (prompt/completion),
  sonst chars/4-Schaetzung. Geloggt pro Tool-Runde im lokalen Fast-Lane.
- log_fast_path(): reiner Skill, 0 Prompt-Tokens — gesparter Claude-Call.
- aggregate() liefert zusaetzlich by_source (calls/tokens_in/tokens_out).
  Alt-Eintraege ohne source zaehlen als claude (rueckwaerts-kompatibel).

Diagnostic Gehirn-Tab: neue Card "Lokales LLM & Claude-Ersparnis" — pro
Fenster (1h/5h/24h/30d) gesparte Claude-Calls (local + fast-path) und
lokale Token-Last (eigene HW, kein Quota) + Info-Block.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
2026-07-11 16:39:02 +02:00
co-authored by Claude Opus 4.8
parent 8bac7c56bc
commit dc043ceb4d
3 changed files with 111 additions and 10 deletions
+13
View File
@@ -30,6 +30,7 @@ from memory import Embedder, VectorStore, MemoryPoint
from prompts import build_system_prompt, IDENTITY_SEED, IDENTITY_ANCHOR, looks_like_identity_break
from proxy_client import ProxyClient, Message as ProxyMessage
import router as router_mod
import metrics
from local_llm import local_llm_chat
import skills as skills_mod
import triggers as triggers_mod
@@ -1220,6 +1221,13 @@ class Agent:
if local_only:
return f"[Lokales LLM nicht erreichbar: {res.get('error', 'unbekannt')}]"
return None
# Metric: dieser lokale Call (echte usage-Tokens wenn der Adapter sie
# liefert). Erfasst pro Tool-Runde — mehrere Runden = mehrere Calls.
try:
metrics.log_local_call(res.get("model") or local_model, messages,
res.get("content") or "", res.get("usage"))
except Exception:
pass
tcs = res.get("tool_calls")
if tcs:
messages.append({"role": "assistant",
@@ -1325,6 +1333,11 @@ class Agent:
self.conversation.add("assistant", fast_reply, project_id=active_project_id)
if active_project_id:
projects_mod.touch_project(active_project_id)
# Metric: Fast-Path spart einen ganzen Claude-Call zum Nulltarif.
try:
metrics.log_fast_path(fast_reply)
except Exception:
pass
# Fast-Path = reiner Steuerbefehl → NICHT vorlesen (speak=False).
# System-Flag statt <voice>-Tag: robust, unabhaengig vom Skill-Inhalt.
return fast_reply, "fast-path", False
+54 -10
View File
@@ -52,25 +52,55 @@ def _messages_tokens(messages: list) -> int:
return total
def log_call(model: str, messages_in: list, reply_text: str = "") -> None:
"""Eine Call-Metric anhaengen. Robust gegen Fehler (silent fail)."""
def _append(model: str, tokens_in: int, tokens_out: int, source: str) -> None:
"""Ein Metric-Entry auf Disk anhaengen. Robust (silent fail)."""
try:
tokens_in = _messages_tokens(messages_in)
tokens_out = _estimate_tokens(reply_text)
line = json.dumps({
"ts": int(time.time() * 1000),
"model": model,
"in": tokens_in,
"out": tokens_out,
"in": int(tokens_in),
"out": int(tokens_out),
"source": source, # "claude" | "local" | "fast-path"
})
METRICS_FILE.parent.mkdir(parents=True, exist_ok=True)
with METRICS_FILE.open("a", encoding="utf-8") as f:
f.write(line + "\n")
# Sanftes Rotate ohne hohe IO-Kosten — nur alle 1000 Calls checken
if (tokens_in + tokens_out) % 1000 < 4:
_maybe_rotate()
except Exception as exc:
logger.warning("metrics.log_call: %s", exc)
logger.warning("metrics._append: %s", exc)
def log_call(model: str, messages_in: list, reply_text: str = "",
source: str = "claude") -> None:
"""Claude-Call-Metric anhaengen (Tokens per chars/4-Schaetzung)."""
_append(model, _messages_tokens(messages_in), _estimate_tokens(reply_text), source)
def log_local_call(model: str, messages_in: list, reply_text: str = "",
usage: dict | None = None) -> None:
"""Lokaler-LLM-Call-Metric. Nutzt echte usage-Tokens (prompt/completion)
wenn der Adapter sie liefert, sonst chars/4-Schaetzung wie bei Claude.
Quelle = 'local' — damit die Ersparnis-Rechnung local von claude trennt."""
tokens_in = tokens_out = None
if isinstance(usage, dict):
pt = usage.get("prompt_tokens")
ct = usage.get("completion_tokens")
if isinstance(pt, (int, float)):
tokens_in = int(pt)
if isinstance(ct, (int, float)):
tokens_out = int(ct)
if tokens_in is None:
tokens_in = _messages_tokens(messages_in)
if tokens_out is None:
tokens_out = _estimate_tokens(reply_text)
_append(model or "local", tokens_in, tokens_out, "local")
def log_fast_path(reply_text: str = "") -> None:
"""Fast-Path (reiner Skill, KEIN LLM) — spart einen ganzen Claude-Call zum
Nulltarif. tokens_in=0 (kein Prompt ans LLM), out = winzige Quittung."""
_append("fast-path", 0, _estimate_tokens(reply_text), "fast-path")
def _maybe_rotate() -> None:
@@ -95,6 +125,11 @@ def aggregate(window_seconds: int) -> dict:
tokens_in = 0
tokens_out = 0
by_model: dict[str, int] = {}
# Aufschluesselung nach Quelle (claude / local / fast-path) fuer die
# Ersparnis-Anzeige im Diagnostic.
def _src_bucket() -> dict:
return {"calls": 0, "tokens_in": 0, "tokens_out": 0}
by_source: dict[str, dict] = {}
if METRICS_FILE.exists():
try:
for raw in METRICS_FILE.read_text(encoding="utf-8").splitlines():
@@ -107,11 +142,19 @@ def aggregate(window_seconds: int) -> dict:
continue
if obj.get("ts", 0) < cutoff_ms:
continue
ti = int(obj.get("in") or 0)
to = int(obj.get("out") or 0)
calls += 1
tokens_in += int(obj.get("in") or 0)
tokens_out += int(obj.get("out") or 0)
tokens_in += ti
tokens_out += to
m = obj.get("model", "?")
by_model[m] = by_model.get(m, 0) + 1
# Alt-Eintraege ohne 'source' zaehlen als claude (Rueckwaerts-Kompat).
src = obj.get("source") or "claude"
b = by_source.setdefault(src, _src_bucket())
b["calls"] += 1
b["tokens_in"] += ti
b["tokens_out"] += to
except Exception as exc:
logger.warning("metrics aggregate: %s", exc)
return {
@@ -120,6 +163,7 @@ def aggregate(window_seconds: int) -> dict:
"tokens_in": tokens_in,
"tokens_out": tokens_out,
"by_model": by_model,
"by_source": by_source,
}