diff --git a/tests/latency_ab.py b/tests/latency_ab.py new file mode 100644 index 0000000..61882d8 --- /dev/null +++ b/tests/latency_ab.py @@ -0,0 +1,131 @@ +"""Ad-hoc TTFT/total latency A/B across Sonnet versions (+ Haiku reference). + +Mirrors the production brain call: Claude Code identity system block, cached +persona, a short voice-style history, one short user turn. Streams to measure +time-to-first-token. Interleaves models each round to cancel network drift. + +Run: .venv/bin/python tests/latency_ab.py +""" +from __future__ import annotations + +import json +import os +import statistics +import time + +import anthropic + +from wsai.backends.claude import ClaudeBrain + +CRED = os.environ.get("CLAUDE_CREDENTIALS_PATH") +if not CRED: + raise SystemExit( + "Set CLAUDE_CREDENTIALS_PATH to your Claude credentials JSON to run this A/B script." + ) +CLAUDE_CODE_ID = "You are Claude Code, Anthropic's official CLI for Claude." + +# (label, model, extra_system_line) — extra line appended to persona to force brevity +BREVITY = "지금부터 답은 무조건 한 문장, 12단어 이내로만. 부연·재확인·군더더기 금지." +VARIANTS = [ + ("sonnet-4-5", "claude-sonnet-4-5", None), + ("sonnet-5+brev", "claude-sonnet-5", BREVITY), + ("haiku-4-5(ref)", "claude-haiku-4-5", None), +] +ROUNDS = 16 + +HISTORY = [ + ("안녕", "[반가움] 안녕! 뭐 하고 있었어?"), + ("그냥 코딩", "[다정] 오 무슨 코딩?"), + ("파이썬", "[신남] 좋네, 잘 되고 있어?"), + ("응 그럭저럭", "[차분] 다행이다."), +] +USER = "지금 몇 시야?" + + +def token() -> str: + with open(CRED) as f: + return json.load(f)["claudeAiOauth"]["accessToken"] + + +def build(extra: str | None = None): + persona = ClaudeBrain.PERSONA + if extra: + persona = persona + "\n\n" + extra + system = [ + {"type": "text", "text": CLAUDE_CODE_ID}, + {"type": "text", "text": persona, "cache_control": {"type": "ephemeral"}}, + ] + msgs = [] + for u, a in HISTORY: + msgs.append({"role": "user", "content": u}) + msgs.append({"role": "assistant", "content": a}) + msgs.append({"role": "user", "content": "[지금 화면] (아직 못 읽음)\n\n" + USER}) + return system, msgs + + +def measure(client, model, system, msgs): + t0 = time.monotonic() + ttft = None + out_tokens = 0 + try: + with client.messages.stream( + model=model, max_tokens=150, system=system, messages=msgs + ) as stream: + for ev in stream.text_stream: + if ttft is None: + ttft = time.monotonic() - t0 + total = time.monotonic() - t0 + final = stream.get_final_message() + out_tokens = final.usage.output_tokens + return ttft, total, out_tokens, None + except Exception as e: # noqa: BLE001 + return None, None, None, f"{type(e).__name__}: {str(e)[:120]}" + + +def main(): + client = anthropic.Anthropic(auth_token=token(), max_retries=1) + prebuilt = {label: build(extra) for label, model, extra in VARIANTS} + results = {label: {"ttft": [], "total": [], "out": []} for label, _, _ in VARIANTS} + dead = set() + # one warm-up per variant to prime connection + cache (excluded from stats) + for label, model, _ in VARIANTS: + system, msgs = prebuilt[label] + _, _, _, err = measure(client, model, system, msgs) + if err: + print(f"[skip] {label} ({model}): {err}") + dead.add(label) + print(f"\nliving: {[l for l, _, _ in VARIANTS if l not in dead]}") + print(f"rounds: {ROUNDS} (interleaved)\n") + for r in range(ROUNDS): + for label, model, _ in VARIANTS: + if label in dead: + continue + system, msgs = prebuilt[label] + ttft, total, out, err = measure(client, model, system, msgs) + if err: + print(f" r{r} {label}: ERR {err}") + continue + results[label]["ttft"].append(ttft) + results[label]["total"].append(total) + results[label]["out"].append(out) + print(f"round {r+1}/{ROUNDS} done") + + print("\n=== median (min–max) over", ROUNDS, "runs ===") + print(f"{'variant':<18} {'TTFT s':<18} {'total s':<18} {'out tok'}") + for label, _, _ in VARIANTS: + d = results[label] + if not d["ttft"]: + print(f"{label:<18} (no data)") + continue + tt = d["ttft"] + to = d["total"] + print( + f"{label:<18} " + f"{statistics.median(tt):.2f} ({min(tt):.2f}-{max(tt):.2f}) " + f"{statistics.median(to):.2f} ({min(to):.2f}-{max(to):.2f}) " + f"{statistics.median(d['out']):.0f}" + ) + + +if __name__ == "__main__": + main() diff --git a/wsai/__main__.py b/wsai/__main__.py index 4ef44c4..7639034 100644 --- a/wsai/__main__.py +++ b/wsai/__main__.py @@ -142,7 +142,7 @@ def _run_voice_server(host: str, port: int) -> None: if os.environ.get("WSAI_BRAIN", "claude").lower() not in ("none", "echo"): try: from .backends.claude import ClaudeBrain - model = os.environ.get("WSAI_BRAIN_MODEL", "claude-sonnet-4-5") + model = os.environ.get("WSAI_BRAIN_MODEL", "claude-sonnet-5") brain = ClaudeBrain(model=model) brain_name = "claude" except Exception as exc: # noqa: BLE001 diff --git a/wsai/backends/claude.py b/wsai/backends/claude.py index e9fe258..d57317f 100644 --- a/wsai/backends/claude.py +++ b/wsai/backends/claude.py @@ -127,6 +127,15 @@ class ClaudeVision: class ClaudeBrain: + # Injected as an always-on, cached system block on top of the (editable) + # persona. Sonnet 5 answers correctly but more verbosely than 4.5 for the + # same voice prompt (measured 46-54 vs 28 output tokens), which erased its + # ~0.4s time-to-first-token advantage in total turn time. This hard brevity + # rule pulls Sonnet 5 back to ~30 tokens, so the faster first token actually + # translates into a faster (and lower-variance) whole reply. Kept separate + # from PERSONA so a dashboard persona edit can never drop it. + BREVITY = "지금부터 답은 무조건 한 문장, 12단어 이내로만. 부연·재확인·군더더기 금지." + PERSONA = ( "너는 디스코드를 이용해 사용자와 실시간으로 대화하는 AI 인공지능이야.\n\n" "1. 역할\n" @@ -163,7 +172,7 @@ class ClaudeBrain: "- 너는 디스코드에서 함께 대화하는 실시간 AI 인공지능이다." ) - def __init__(self, *, model: str = "claude-sonnet-4-5", api_key: str | None = None) -> None: + def __init__(self, *, model: str = "claude-sonnet-5", api_key: str | None = None) -> None: self.model = model self._auth = _Auth(api_key) @@ -176,8 +185,9 @@ class ClaudeBrain: msgs.append({"role": "user", "content": screen_note + user_text}) client = self._auth.client() # Read the persona live each turn so a dashboard edit applies immediately - # (falls back to the built-in PERSONA when no override is saved). - system = self._auth.system(get_persona(self.PERSONA)) + # (falls back to the built-in PERSONA when no override is saved). The + # brevity rule is appended as its own block so it survives persona edits. + system = self._auth.system(get_persona(self.PERSONA), self.BREVITY) if system: # Cache the (static) system prompt so repeat turns skip re-processing # it — lower time-to-first-token. No-op below the model's cache diff --git a/wsai/config.py b/wsai/config.py index bf3aaa9..397b8b7 100644 --- a/wsai/config.py +++ b/wsai/config.py @@ -24,7 +24,7 @@ class Settings: text: str | None = None # None | (discord) capture_interval: float = 1.5 - anthropic_model: str = "claude-sonnet-4-5" + anthropic_model: str = "claude-sonnet-5" @classmethod def from_env(cls) -> "Settings": diff --git a/wsai/dashboard.py b/wsai/dashboard.py index 60e0f33..38452d4 100644 --- a/wsai/dashboard.py +++ b/wsai/dashboard.py @@ -649,7 +649,7 @@ class Dashboard: # -- live model switching (STT size / LLM model) --------------------- # STT_OPTIONS = ["tiny", "base", "small", "medium", "large-v3"] - LLM_OPTIONS = ["claude-haiku-4-5", "claude-sonnet-4-5"] + LLM_OPTIONS = ["claude-haiku-4-5", "claude-sonnet-4-5", "claude-sonnet-5"] def models_settings(self) -> dict: stt = self.stt @@ -1193,7 +1193,7 @@ PAGE = r""" -

STT는 전환 시 모델을 다시 로드합니다(처음 medium/large-v3는 다운로드로 수 분 걸릴 수 있어요). LLM은 다음 답변부터 즉시 적용됩니다. 서비스 재시작 시 기본값(medium · Haiku)으로 돌아갑니다.

+

STT는 전환 시 모델을 다시 로드합니다(처음 medium/large-v3는 다운로드로 수 분 걸릴 수 있어요). LLM은 다음 답변부터 즉시 적용됩니다. 마지막으로 고른 모델은 재시작 후에도 유지됩니다(기본값 Sonnet 5).

@@ -1519,7 +1519,8 @@ function wireCollapse(toggleId, bodyId, caretId){ const STT_LABEL = {tiny:'tiny (가장 빠름)', base:'base', small:'small (빠름)', medium:'medium (기본·정확)', 'large-v3':'large-v3 (최고 정확도)'}; const LLM_LABEL = {'claude-haiku-4-5':'Haiku 4.5 (가장 빠름)', - 'claude-sonnet-4-5':'Sonnet 4.5 (고품질·조금 느림)'}; + 'claude-sonnet-4-5':'Sonnet 4.5 (고품질·조금 느림)', + 'claude-sonnet-5':'Sonnet 5 (기본 · 빠르고 고품질)'}; function fillSel(sel, options, current, labels){ sel.innerHTML=''; options.forEach(o=>{ const el=document.createElement('option');