"""Ad-hoc TTFT/total latency A/B across Sonnet versions (+ Haiku reference). Mirrors the production brain call: Claude Code identity system block, cached persona, a short voice-style history, one short user turn. Streams to measure time-to-first-token. Interleaves models each round to cancel network drift. Run: .venv/bin/python tests/latency_ab.py """ from __future__ import annotations import json import os import statistics import time import anthropic from wsai.backends.claude import ClaudeBrain CRED = os.environ.get("CLAUDE_CREDENTIALS_PATH") if not CRED: raise SystemExit( "Set CLAUDE_CREDENTIALS_PATH to your Claude credentials JSON to run this A/B script." ) CLAUDE_CODE_ID = "You are Claude Code, Anthropic's official CLI for Claude." # (label, model, extra_system_line) — extra line appended to persona to force brevity BREVITY = "지금부터 답은 무조건 한 문장, 12단어 이내로만. 부연·재확인·군더더기 금지." VARIANTS = [ ("sonnet-4-5", "claude-sonnet-4-5", None), ("sonnet-5+brev", "claude-sonnet-5", BREVITY), ("haiku-4-5(ref)", "claude-haiku-4-5", None), ] ROUNDS = 16 HISTORY = [ ("안녕", "[반가움] 안녕! 뭐 하고 있었어?"), ("그냥 코딩", "[다정] 오 무슨 코딩?"), ("파이썬", "[신남] 좋네, 잘 되고 있어?"), ("응 그럭저럭", "[차분] 다행이다."), ] USER = "지금 몇 시야?" def token() -> str: with open(CRED) as f: return json.load(f)["claudeAiOauth"]["accessToken"] def build(extra: str | None = None): persona = ClaudeBrain.PERSONA if extra: persona = persona + "\n\n" + extra system = [ {"type": "text", "text": CLAUDE_CODE_ID}, {"type": "text", "text": persona, "cache_control": {"type": "ephemeral"}}, ] msgs = [] for u, a in HISTORY: msgs.append({"role": "user", "content": u}) msgs.append({"role": "assistant", "content": a}) msgs.append({"role": "user", "content": "[지금 화면] (아직 못 읽음)\n\n" + USER}) return system, msgs def measure(client, model, system, msgs): t0 = time.monotonic() ttft = None out_tokens = 0 try: with client.messages.stream( model=model, max_tokens=150, system=system, messages=msgs ) as stream: for ev in stream.text_stream: if ttft is None: ttft = time.monotonic() - t0 total = time.monotonic() - t0 final = stream.get_final_message() out_tokens = final.usage.output_tokens return ttft, total, out_tokens, None except Exception as e: # noqa: BLE001 return None, None, None, f"{type(e).__name__}: {str(e)[:120]}" def main(): client = anthropic.Anthropic(auth_token=token(), max_retries=1) prebuilt = {label: build(extra) for label, model, extra in VARIANTS} results = {label: {"ttft": [], "total": [], "out": []} for label, _, _ in VARIANTS} dead = set() # one warm-up per variant to prime connection + cache (excluded from stats) for label, model, _ in VARIANTS: system, msgs = prebuilt[label] _, _, _, err = measure(client, model, system, msgs) if err: print(f"[skip] {label} ({model}): {err}") dead.add(label) print(f"\nliving: {[l for l, _, _ in VARIANTS if l not in dead]}") print(f"rounds: {ROUNDS} (interleaved)\n") for r in range(ROUNDS): for label, model, _ in VARIANTS: if label in dead: continue system, msgs = prebuilt[label] ttft, total, out, err = measure(client, model, system, msgs) if err: print(f" r{r} {label}: ERR {err}") continue results[label]["ttft"].append(ttft) results[label]["total"].append(total) results[label]["out"].append(out) print(f"round {r+1}/{ROUNDS} done") print("\n=== median (min–max) over", ROUNDS, "runs ===") print(f"{'variant':<18} {'TTFT s':<18} {'total s':<18} {'out tok'}") for label, _, _ in VARIANTS: d = results[label] if not d["ttft"]: print(f"{label:<18} (no data)") continue tt = d["ttft"] to = d["total"] print( f"{label:<18} " f"{statistics.median(tt):.2f} ({min(tt):.2f}-{max(tt):.2f}) " f"{statistics.median(to):.2f} ({min(to):.2f}-{max(to):.2f}) " f"{statistics.median(d['out']):.0f}" ) if __name__ == "__main__": main()