Sonnet 5 reaches first token ~0.4s sooner than Sonnet 4.5 but answers the same voice prompt more verbosely (measured 46-54 vs ~28 output tokens), which erased the win in total turn time. Add a cached, persona-independent brevity system block that pulls output back to ~30 tokens, so the faster first token becomes a faster, lower-variance whole reply. Measured (16-round interleaved A/B, production-shaped call): sonnet-4-5 TTFT 1.26s total 2.02s (tail 3.26s) out 29 sonnet-5+brev TTFT 0.85s total 1.70s (tail 2.28s) out 30 - ClaudeBrain: default model claude-sonnet-5 + BREVITY block (cached with the persona prefix so a dashboard persona edit can't drop it). - Defaults aligned: config.Settings.anthropic_model and the voice-server WSAI_BRAIN_MODEL default -> claude-sonnet-5. - Dashboard: add claude-sonnet-5 to LLM_OPTIONS + JS label; fix stale restart hint. - tests/latency_ab.py: reproducible model-latency A/B harness (reads CLAUDE_CREDENTIALS_PATH; makes live API calls, so not a pytest test). Streaming TTS was intentionally not added: this bot answers in one sentence, where sentence-level streaming has no overlap to exploit, and it would require rearchitecting both the Python endpoint and the node playback. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
132 lines
4.5 KiB
Python
132 lines
4.5 KiB
Python
"""Ad-hoc TTFT/total latency A/B across Sonnet versions (+ Haiku reference).
|
||
|
||
Mirrors the production brain call: Claude Code identity system block, cached
|
||
persona, a short voice-style history, one short user turn. Streams to measure
|
||
time-to-first-token. Interleaves models each round to cancel network drift.
|
||
|
||
Run: .venv/bin/python tests/latency_ab.py
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import json
|
||
import os
|
||
import statistics
|
||
import time
|
||
|
||
import anthropic
|
||
|
||
from wsai.backends.claude import ClaudeBrain
|
||
|
||
CRED = os.environ.get("CLAUDE_CREDENTIALS_PATH")
|
||
if not CRED:
|
||
raise SystemExit(
|
||
"Set CLAUDE_CREDENTIALS_PATH to your Claude credentials JSON to run this A/B script."
|
||
)
|
||
CLAUDE_CODE_ID = "You are Claude Code, Anthropic's official CLI for Claude."
|
||
|
||
# (label, model, extra_system_line) — extra line appended to persona to force brevity
|
||
BREVITY = "지금부터 답은 무조건 한 문장, 12단어 이내로만. 부연·재확인·군더더기 금지."
|
||
VARIANTS = [
|
||
("sonnet-4-5", "claude-sonnet-4-5", None),
|
||
("sonnet-5+brev", "claude-sonnet-5", BREVITY),
|
||
("haiku-4-5(ref)", "claude-haiku-4-5", None),
|
||
]
|
||
ROUNDS = 16
|
||
|
||
HISTORY = [
|
||
("안녕", "[반가움] 안녕! 뭐 하고 있었어?"),
|
||
("그냥 코딩", "[다정] 오 무슨 코딩?"),
|
||
("파이썬", "[신남] 좋네, 잘 되고 있어?"),
|
||
("응 그럭저럭", "[차분] 다행이다."),
|
||
]
|
||
USER = "지금 몇 시야?"
|
||
|
||
|
||
def token() -> str:
|
||
with open(CRED) as f:
|
||
return json.load(f)["claudeAiOauth"]["accessToken"]
|
||
|
||
|
||
def build(extra: str | None = None):
|
||
persona = ClaudeBrain.PERSONA
|
||
if extra:
|
||
persona = persona + "\n\n" + extra
|
||
system = [
|
||
{"type": "text", "text": CLAUDE_CODE_ID},
|
||
{"type": "text", "text": persona, "cache_control": {"type": "ephemeral"}},
|
||
]
|
||
msgs = []
|
||
for u, a in HISTORY:
|
||
msgs.append({"role": "user", "content": u})
|
||
msgs.append({"role": "assistant", "content": a})
|
||
msgs.append({"role": "user", "content": "[지금 화면] (아직 못 읽음)\n\n" + USER})
|
||
return system, msgs
|
||
|
||
|
||
def measure(client, model, system, msgs):
|
||
t0 = time.monotonic()
|
||
ttft = None
|
||
out_tokens = 0
|
||
try:
|
||
with client.messages.stream(
|
||
model=model, max_tokens=150, system=system, messages=msgs
|
||
) as stream:
|
||
for ev in stream.text_stream:
|
||
if ttft is None:
|
||
ttft = time.monotonic() - t0
|
||
total = time.monotonic() - t0
|
||
final = stream.get_final_message()
|
||
out_tokens = final.usage.output_tokens
|
||
return ttft, total, out_tokens, None
|
||
except Exception as e: # noqa: BLE001
|
||
return None, None, None, f"{type(e).__name__}: {str(e)[:120]}"
|
||
|
||
|
||
def main():
|
||
client = anthropic.Anthropic(auth_token=token(), max_retries=1)
|
||
prebuilt = {label: build(extra) for label, model, extra in VARIANTS}
|
||
results = {label: {"ttft": [], "total": [], "out": []} for label, _, _ in VARIANTS}
|
||
dead = set()
|
||
# one warm-up per variant to prime connection + cache (excluded from stats)
|
||
for label, model, _ in VARIANTS:
|
||
system, msgs = prebuilt[label]
|
||
_, _, _, err = measure(client, model, system, msgs)
|
||
if err:
|
||
print(f"[skip] {label} ({model}): {err}")
|
||
dead.add(label)
|
||
print(f"\nliving: {[l for l, _, _ in VARIANTS if l not in dead]}")
|
||
print(f"rounds: {ROUNDS} (interleaved)\n")
|
||
for r in range(ROUNDS):
|
||
for label, model, _ in VARIANTS:
|
||
if label in dead:
|
||
continue
|
||
system, msgs = prebuilt[label]
|
||
ttft, total, out, err = measure(client, model, system, msgs)
|
||
if err:
|
||
print(f" r{r} {label}: ERR {err}")
|
||
continue
|
||
results[label]["ttft"].append(ttft)
|
||
results[label]["total"].append(total)
|
||
results[label]["out"].append(out)
|
||
print(f"round {r+1}/{ROUNDS} done")
|
||
|
||
print("\n=== median (min–max) over", ROUNDS, "runs ===")
|
||
print(f"{'variant':<18} {'TTFT s':<18} {'total s':<18} {'out tok'}")
|
||
for label, _, _ in VARIANTS:
|
||
d = results[label]
|
||
if not d["ttft"]:
|
||
print(f"{label:<18} (no data)")
|
||
continue
|
||
tt = d["ttft"]
|
||
to = d["total"]
|
||
print(
|
||
f"{label:<18} "
|
||
f"{statistics.median(tt):.2f} ({min(tt):.2f}-{max(tt):.2f}) "
|
||
f"{statistics.median(to):.2f} ({min(to):.2f}-{max(to):.2f}) "
|
||
f"{statistics.median(d['out']):.0f}"
|
||
)
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|