perf(brain): switch voice brain to Sonnet 5 with an always-on brevity rule
Sonnet 5 reaches first token ~0.4s sooner than Sonnet 4.5 but answers the same voice prompt more verbosely (measured 46-54 vs ~28 output tokens), which erased the win in total turn time. Add a cached, persona-independent brevity system block that pulls output back to ~30 tokens, so the faster first token becomes a faster, lower-variance whole reply. Measured (16-round interleaved A/B, production-shaped call): sonnet-4-5 TTFT 1.26s total 2.02s (tail 3.26s) out 29 sonnet-5+brev TTFT 0.85s total 1.70s (tail 2.28s) out 30 - ClaudeBrain: default model claude-sonnet-5 + BREVITY block (cached with the persona prefix so a dashboard persona edit can't drop it). - Defaults aligned: config.Settings.anthropic_model and the voice-server WSAI_BRAIN_MODEL default -> claude-sonnet-5. - Dashboard: add claude-sonnet-5 to LLM_OPTIONS + JS label; fix stale restart hint. - tests/latency_ab.py: reproducible model-latency A/B harness (reads CLAUDE_CREDENTIALS_PATH; makes live API calls, so not a pytest test). Streaming TTS was intentionally not added: this bot answers in one sentence, where sentence-level streaming has no overlap to exploit, and it would require rearchitecting both the Python endpoint and the node playback. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
131
tests/latency_ab.py
Normal file
131
tests/latency_ab.py
Normal file
@@ -0,0 +1,131 @@
|
|||||||
|
"""Ad-hoc TTFT/total latency A/B across Sonnet versions (+ Haiku reference).
|
||||||
|
|
||||||
|
Mirrors the production brain call: Claude Code identity system block, cached
|
||||||
|
persona, a short voice-style history, one short user turn. Streams to measure
|
||||||
|
time-to-first-token. Interleaves models each round to cancel network drift.
|
||||||
|
|
||||||
|
Run: .venv/bin/python tests/latency_ab.py
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import statistics
|
||||||
|
import time
|
||||||
|
|
||||||
|
import anthropic
|
||||||
|
|
||||||
|
from wsai.backends.claude import ClaudeBrain
|
||||||
|
|
||||||
|
CRED = os.environ.get("CLAUDE_CREDENTIALS_PATH")
|
||||||
|
if not CRED:
|
||||||
|
raise SystemExit(
|
||||||
|
"Set CLAUDE_CREDENTIALS_PATH to your Claude credentials JSON to run this A/B script."
|
||||||
|
)
|
||||||
|
CLAUDE_CODE_ID = "You are Claude Code, Anthropic's official CLI for Claude."
|
||||||
|
|
||||||
|
# (label, model, extra_system_line) — extra line appended to persona to force brevity
|
||||||
|
BREVITY = "지금부터 답은 무조건 한 문장, 12단어 이내로만. 부연·재확인·군더더기 금지."
|
||||||
|
VARIANTS = [
|
||||||
|
("sonnet-4-5", "claude-sonnet-4-5", None),
|
||||||
|
("sonnet-5+brev", "claude-sonnet-5", BREVITY),
|
||||||
|
("haiku-4-5(ref)", "claude-haiku-4-5", None),
|
||||||
|
]
|
||||||
|
ROUNDS = 16
|
||||||
|
|
||||||
|
HISTORY = [
|
||||||
|
("안녕", "[반가움] 안녕! 뭐 하고 있었어?"),
|
||||||
|
("그냥 코딩", "[다정] 오 무슨 코딩?"),
|
||||||
|
("파이썬", "[신남] 좋네, 잘 되고 있어?"),
|
||||||
|
("응 그럭저럭", "[차분] 다행이다."),
|
||||||
|
]
|
||||||
|
USER = "지금 몇 시야?"
|
||||||
|
|
||||||
|
|
||||||
|
def token() -> str:
|
||||||
|
with open(CRED) as f:
|
||||||
|
return json.load(f)["claudeAiOauth"]["accessToken"]
|
||||||
|
|
||||||
|
|
||||||
|
def build(extra: str | None = None):
|
||||||
|
persona = ClaudeBrain.PERSONA
|
||||||
|
if extra:
|
||||||
|
persona = persona + "\n\n" + extra
|
||||||
|
system = [
|
||||||
|
{"type": "text", "text": CLAUDE_CODE_ID},
|
||||||
|
{"type": "text", "text": persona, "cache_control": {"type": "ephemeral"}},
|
||||||
|
]
|
||||||
|
msgs = []
|
||||||
|
for u, a in HISTORY:
|
||||||
|
msgs.append({"role": "user", "content": u})
|
||||||
|
msgs.append({"role": "assistant", "content": a})
|
||||||
|
msgs.append({"role": "user", "content": "[지금 화면] (아직 못 읽음)\n\n" + USER})
|
||||||
|
return system, msgs
|
||||||
|
|
||||||
|
|
||||||
|
def measure(client, model, system, msgs):
|
||||||
|
t0 = time.monotonic()
|
||||||
|
ttft = None
|
||||||
|
out_tokens = 0
|
||||||
|
try:
|
||||||
|
with client.messages.stream(
|
||||||
|
model=model, max_tokens=150, system=system, messages=msgs
|
||||||
|
) as stream:
|
||||||
|
for ev in stream.text_stream:
|
||||||
|
if ttft is None:
|
||||||
|
ttft = time.monotonic() - t0
|
||||||
|
total = time.monotonic() - t0
|
||||||
|
final = stream.get_final_message()
|
||||||
|
out_tokens = final.usage.output_tokens
|
||||||
|
return ttft, total, out_tokens, None
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
return None, None, None, f"{type(e).__name__}: {str(e)[:120]}"
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
client = anthropic.Anthropic(auth_token=token(), max_retries=1)
|
||||||
|
prebuilt = {label: build(extra) for label, model, extra in VARIANTS}
|
||||||
|
results = {label: {"ttft": [], "total": [], "out": []} for label, _, _ in VARIANTS}
|
||||||
|
dead = set()
|
||||||
|
# one warm-up per variant to prime connection + cache (excluded from stats)
|
||||||
|
for label, model, _ in VARIANTS:
|
||||||
|
system, msgs = prebuilt[label]
|
||||||
|
_, _, _, err = measure(client, model, system, msgs)
|
||||||
|
if err:
|
||||||
|
print(f"[skip] {label} ({model}): {err}")
|
||||||
|
dead.add(label)
|
||||||
|
print(f"\nliving: {[l for l, _, _ in VARIANTS if l not in dead]}")
|
||||||
|
print(f"rounds: {ROUNDS} (interleaved)\n")
|
||||||
|
for r in range(ROUNDS):
|
||||||
|
for label, model, _ in VARIANTS:
|
||||||
|
if label in dead:
|
||||||
|
continue
|
||||||
|
system, msgs = prebuilt[label]
|
||||||
|
ttft, total, out, err = measure(client, model, system, msgs)
|
||||||
|
if err:
|
||||||
|
print(f" r{r} {label}: ERR {err}")
|
||||||
|
continue
|
||||||
|
results[label]["ttft"].append(ttft)
|
||||||
|
results[label]["total"].append(total)
|
||||||
|
results[label]["out"].append(out)
|
||||||
|
print(f"round {r+1}/{ROUNDS} done")
|
||||||
|
|
||||||
|
print("\n=== median (min–max) over", ROUNDS, "runs ===")
|
||||||
|
print(f"{'variant':<18} {'TTFT s':<18} {'total s':<18} {'out tok'}")
|
||||||
|
for label, _, _ in VARIANTS:
|
||||||
|
d = results[label]
|
||||||
|
if not d["ttft"]:
|
||||||
|
print(f"{label:<18} (no data)")
|
||||||
|
continue
|
||||||
|
tt = d["ttft"]
|
||||||
|
to = d["total"]
|
||||||
|
print(
|
||||||
|
f"{label:<18} "
|
||||||
|
f"{statistics.median(tt):.2f} ({min(tt):.2f}-{max(tt):.2f}) "
|
||||||
|
f"{statistics.median(to):.2f} ({min(to):.2f}-{max(to):.2f}) "
|
||||||
|
f"{statistics.median(d['out']):.0f}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -142,7 +142,7 @@ def _run_voice_server(host: str, port: int) -> None:
|
|||||||
if os.environ.get("WSAI_BRAIN", "claude").lower() not in ("none", "echo"):
|
if os.environ.get("WSAI_BRAIN", "claude").lower() not in ("none", "echo"):
|
||||||
try:
|
try:
|
||||||
from .backends.claude import ClaudeBrain
|
from .backends.claude import ClaudeBrain
|
||||||
model = os.environ.get("WSAI_BRAIN_MODEL", "claude-sonnet-4-5")
|
model = os.environ.get("WSAI_BRAIN_MODEL", "claude-sonnet-5")
|
||||||
brain = ClaudeBrain(model=model)
|
brain = ClaudeBrain(model=model)
|
||||||
brain_name = "claude"
|
brain_name = "claude"
|
||||||
except Exception as exc: # noqa: BLE001
|
except Exception as exc: # noqa: BLE001
|
||||||
|
|||||||
@@ -127,6 +127,15 @@ class ClaudeVision:
|
|||||||
|
|
||||||
|
|
||||||
class ClaudeBrain:
|
class ClaudeBrain:
|
||||||
|
# Injected as an always-on, cached system block on top of the (editable)
|
||||||
|
# persona. Sonnet 5 answers correctly but more verbosely than 4.5 for the
|
||||||
|
# same voice prompt (measured 46-54 vs 28 output tokens), which erased its
|
||||||
|
# ~0.4s time-to-first-token advantage in total turn time. This hard brevity
|
||||||
|
# rule pulls Sonnet 5 back to ~30 tokens, so the faster first token actually
|
||||||
|
# translates into a faster (and lower-variance) whole reply. Kept separate
|
||||||
|
# from PERSONA so a dashboard persona edit can never drop it.
|
||||||
|
BREVITY = "지금부터 답은 무조건 한 문장, 12단어 이내로만. 부연·재확인·군더더기 금지."
|
||||||
|
|
||||||
PERSONA = (
|
PERSONA = (
|
||||||
"너는 디스코드를 이용해 사용자와 실시간으로 대화하는 AI 인공지능이야.\n\n"
|
"너는 디스코드를 이용해 사용자와 실시간으로 대화하는 AI 인공지능이야.\n\n"
|
||||||
"1. 역할\n"
|
"1. 역할\n"
|
||||||
@@ -163,7 +172,7 @@ class ClaudeBrain:
|
|||||||
"- 너는 디스코드에서 함께 대화하는 실시간 AI 인공지능이다."
|
"- 너는 디스코드에서 함께 대화하는 실시간 AI 인공지능이다."
|
||||||
)
|
)
|
||||||
|
|
||||||
def __init__(self, *, model: str = "claude-sonnet-4-5", api_key: str | None = None) -> None:
|
def __init__(self, *, model: str = "claude-sonnet-5", api_key: str | None = None) -> None:
|
||||||
self.model = model
|
self.model = model
|
||||||
self._auth = _Auth(api_key)
|
self._auth = _Auth(api_key)
|
||||||
|
|
||||||
@@ -176,8 +185,9 @@ class ClaudeBrain:
|
|||||||
msgs.append({"role": "user", "content": screen_note + user_text})
|
msgs.append({"role": "user", "content": screen_note + user_text})
|
||||||
client = self._auth.client()
|
client = self._auth.client()
|
||||||
# Read the persona live each turn so a dashboard edit applies immediately
|
# Read the persona live each turn so a dashboard edit applies immediately
|
||||||
# (falls back to the built-in PERSONA when no override is saved).
|
# (falls back to the built-in PERSONA when no override is saved). The
|
||||||
system = self._auth.system(get_persona(self.PERSONA))
|
# brevity rule is appended as its own block so it survives persona edits.
|
||||||
|
system = self._auth.system(get_persona(self.PERSONA), self.BREVITY)
|
||||||
if system:
|
if system:
|
||||||
# Cache the (static) system prompt so repeat turns skip re-processing
|
# Cache the (static) system prompt so repeat turns skip re-processing
|
||||||
# it — lower time-to-first-token. No-op below the model's cache
|
# it — lower time-to-first-token. No-op below the model's cache
|
||||||
|
|||||||
@@ -24,7 +24,7 @@ class Settings:
|
|||||||
text: str | None = None # None | (discord)
|
text: str | None = None # None | (discord)
|
||||||
|
|
||||||
capture_interval: float = 1.5
|
capture_interval: float = 1.5
|
||||||
anthropic_model: str = "claude-sonnet-4-5"
|
anthropic_model: str = "claude-sonnet-5"
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def from_env(cls) -> "Settings":
|
def from_env(cls) -> "Settings":
|
||||||
|
|||||||
@@ -649,7 +649,7 @@ class Dashboard:
|
|||||||
|
|
||||||
# -- live model switching (STT size / LLM model) --------------------- #
|
# -- live model switching (STT size / LLM model) --------------------- #
|
||||||
STT_OPTIONS = ["tiny", "base", "small", "medium", "large-v3"]
|
STT_OPTIONS = ["tiny", "base", "small", "medium", "large-v3"]
|
||||||
LLM_OPTIONS = ["claude-haiku-4-5", "claude-sonnet-4-5"]
|
LLM_OPTIONS = ["claude-haiku-4-5", "claude-sonnet-4-5", "claude-sonnet-5"]
|
||||||
|
|
||||||
def models_settings(self) -> dict:
|
def models_settings(self) -> dict:
|
||||||
stt = self.stt
|
stt = self.stt
|
||||||
@@ -1193,7 +1193,7 @@ PAGE = r"""<!DOCTYPE html>
|
|||||||
<div id="mLoad" class="mload" style="display:none">
|
<div id="mLoad" class="mload" style="display:none">
|
||||||
<span class="spin"></span><span id="mLoadText">로딩 중…</span>
|
<span class="spin"></span><span id="mLoadText">로딩 중…</span>
|
||||||
</div>
|
</div>
|
||||||
<p class="cfghint" style="margin:8px 0 0">STT는 전환 시 모델을 다시 로드합니다(처음 medium/large-v3는 다운로드로 수 분 걸릴 수 있어요). LLM은 다음 답변부터 즉시 적용됩니다. 서비스 재시작 시 기본값(medium · Haiku)으로 돌아갑니다.</p>
|
<p class="cfghint" style="margin:8px 0 0">STT는 전환 시 모델을 다시 로드합니다(처음 medium/large-v3는 다운로드로 수 분 걸릴 수 있어요). LLM은 다음 답변부터 즉시 적용됩니다. 마지막으로 고른 모델은 재시작 후에도 유지됩니다(기본값 Sonnet 5).</p>
|
||||||
</div>
|
</div>
|
||||||
</section>
|
</section>
|
||||||
<div class="demobar" id="demobar" style="display:none"></div>
|
<div class="demobar" id="demobar" style="display:none"></div>
|
||||||
@@ -1519,7 +1519,8 @@ function wireCollapse(toggleId, bodyId, caretId){
|
|||||||
const STT_LABEL = {tiny:'tiny (가장 빠름)', base:'base', small:'small (빠름)',
|
const STT_LABEL = {tiny:'tiny (가장 빠름)', base:'base', small:'small (빠름)',
|
||||||
medium:'medium (기본·정확)', 'large-v3':'large-v3 (최고 정확도)'};
|
medium:'medium (기본·정확)', 'large-v3':'large-v3 (최고 정확도)'};
|
||||||
const LLM_LABEL = {'claude-haiku-4-5':'Haiku 4.5 (가장 빠름)',
|
const LLM_LABEL = {'claude-haiku-4-5':'Haiku 4.5 (가장 빠름)',
|
||||||
'claude-sonnet-4-5':'Sonnet 4.5 (고품질·조금 느림)'};
|
'claude-sonnet-4-5':'Sonnet 4.5 (고품질·조금 느림)',
|
||||||
|
'claude-sonnet-5':'Sonnet 5 (기본 · 빠르고 고품질)'};
|
||||||
function fillSel(sel, options, current, labels){
|
function fillSel(sel, options, current, labels){
|
||||||
sel.innerHTML='';
|
sel.innerHTML='';
|
||||||
options.forEach(o=>{ const el=document.createElement('option');
|
options.forEach(o=>{ const el=document.createElement('option');
|
||||||
|
|||||||
Reference in New Issue
Block a user