Files
watch_sceen_ai/tests/latency_ab.py
EJClaw 04664ce61a perf(brain): switch voice brain to Sonnet 5 with an always-on brevity rule
Sonnet 5 reaches first token ~0.4s sooner than Sonnet 4.5 but answers the
same voice prompt more verbosely (measured 46-54 vs ~28 output tokens),
which erased the win in total turn time. Add a cached, persona-independent
brevity system block that pulls output back to ~30 tokens, so the faster
first token becomes a faster, lower-variance whole reply.

Measured (16-round interleaved A/B, production-shaped call):
  sonnet-4-5      TTFT 1.26s  total 2.02s (tail 3.26s)  out 29
  sonnet-5+brev   TTFT 0.85s  total 1.70s (tail 2.28s)  out 30

- ClaudeBrain: default model claude-sonnet-5 + BREVITY block (cached with
  the persona prefix so a dashboard persona edit can't drop it).
- Defaults aligned: config.Settings.anthropic_model and the voice-server
  WSAI_BRAIN_MODEL default -> claude-sonnet-5.
- Dashboard: add claude-sonnet-5 to LLM_OPTIONS + JS label; fix stale
  restart hint.
- tests/latency_ab.py: reproducible model-latency A/B harness (reads
  CLAUDE_CREDENTIALS_PATH; makes live API calls, so not a pytest test).

Streaming TTS was intentionally not added: this bot answers in one sentence,
where sentence-level streaming has no overlap to exploit, and it would
require rearchitecting both the Python endpoint and the node playback.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-08-28 20:57:01 +09:00

132 lines
4.5 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Ad-hoc TTFT/total latency A/B across Sonnet versions (+ Haiku reference).
Mirrors the production brain call: Claude Code identity system block, cached
persona, a short voice-style history, one short user turn. Streams to measure
time-to-first-token. Interleaves models each round to cancel network drift.
Run: .venv/bin/python tests/latency_ab.py
"""
from __future__ import annotations
import json
import os
import statistics
import time
import anthropic
from wsai.backends.claude import ClaudeBrain
CRED = os.environ.get("CLAUDE_CREDENTIALS_PATH")
if not CRED:
raise SystemExit(
"Set CLAUDE_CREDENTIALS_PATH to your Claude credentials JSON to run this A/B script."
)
CLAUDE_CODE_ID = "You are Claude Code, Anthropic's official CLI for Claude."
# (label, model, extra_system_line) — extra line appended to persona to force brevity
BREVITY = "지금부터 답은 무조건 한 문장, 12단어 이내로만. 부연·재확인·군더더기 금지."
VARIANTS = [
("sonnet-4-5", "claude-sonnet-4-5", None),
("sonnet-5+brev", "claude-sonnet-5", BREVITY),
("haiku-4-5(ref)", "claude-haiku-4-5", None),
]
ROUNDS = 16
HISTORY = [
("안녕", "[반가움] 안녕! 뭐 하고 있었어?"),
("그냥 코딩", "[다정] 오 무슨 코딩?"),
("파이썬", "[신남] 좋네, 잘 되고 있어?"),
("응 그럭저럭", "[차분] 다행이다."),
]
USER = "지금 몇 시야?"
def token() -> str:
with open(CRED) as f:
return json.load(f)["claudeAiOauth"]["accessToken"]
def build(extra: str | None = None):
persona = ClaudeBrain.PERSONA
if extra:
persona = persona + "\n\n" + extra
system = [
{"type": "text", "text": CLAUDE_CODE_ID},
{"type": "text", "text": persona, "cache_control": {"type": "ephemeral"}},
]
msgs = []
for u, a in HISTORY:
msgs.append({"role": "user", "content": u})
msgs.append({"role": "assistant", "content": a})
msgs.append({"role": "user", "content": "[지금 화면] (아직 못 읽음)\n\n" + USER})
return system, msgs
def measure(client, model, system, msgs):
t0 = time.monotonic()
ttft = None
out_tokens = 0
try:
with client.messages.stream(
model=model, max_tokens=150, system=system, messages=msgs
) as stream:
for ev in stream.text_stream:
if ttft is None:
ttft = time.monotonic() - t0
total = time.monotonic() - t0
final = stream.get_final_message()
out_tokens = final.usage.output_tokens
return ttft, total, out_tokens, None
except Exception as e: # noqa: BLE001
return None, None, None, f"{type(e).__name__}: {str(e)[:120]}"
def main():
client = anthropic.Anthropic(auth_token=token(), max_retries=1)
prebuilt = {label: build(extra) for label, model, extra in VARIANTS}
results = {label: {"ttft": [], "total": [], "out": []} for label, _, _ in VARIANTS}
dead = set()
# one warm-up per variant to prime connection + cache (excluded from stats)
for label, model, _ in VARIANTS:
system, msgs = prebuilt[label]
_, _, _, err = measure(client, model, system, msgs)
if err:
print(f"[skip] {label} ({model}): {err}")
dead.add(label)
print(f"\nliving: {[l for l, _, _ in VARIANTS if l not in dead]}")
print(f"rounds: {ROUNDS} (interleaved)\n")
for r in range(ROUNDS):
for label, model, _ in VARIANTS:
if label in dead:
continue
system, msgs = prebuilt[label]
ttft, total, out, err = measure(client, model, system, msgs)
if err:
print(f" r{r} {label}: ERR {err}")
continue
results[label]["ttft"].append(ttft)
results[label]["total"].append(total)
results[label]["out"].append(out)
print(f"round {r+1}/{ROUNDS} done")
print("\n=== median (min–max) over", ROUNDS, "runs ===")
print(f"{'variant':<18} {'TTFT s':<18} {'total s':<18} {'out tok'}")
for label, _, _ in VARIANTS:
d = results[label]
if not d["ttft"]:
print(f"{label:<18} (no data)")
continue
tt = d["ttft"]
to = d["total"]
print(
f"{label:<18} "
f"{statistics.median(tt):.2f} ({min(tt):.2f}-{max(tt):.2f}) "
f"{statistics.median(to):.2f} ({min(to):.2f}-{max(to):.2f}) "
f"{statistics.median(d['out']):.0f}"
)
if __name__ == "__main__":
main()