Files
watch_sceen_ai/wsai/backends/claude.py
EJClaw 04664ce61a perf(brain): switch voice brain to Sonnet 5 with an always-on brevity rule
Sonnet 5 reaches first token ~0.4s sooner than Sonnet 4.5 but answers the
same voice prompt more verbosely (measured 46-54 vs ~28 output tokens),
which erased the win in total turn time. Add a cached, persona-independent
brevity system block that pulls output back to ~30 tokens, so the faster
first token becomes a faster, lower-variance whole reply.

Measured (16-round interleaved A/B, production-shaped call):
  sonnet-4-5      TTFT 1.26s  total 2.02s (tail 3.26s)  out 29
  sonnet-5+brev   TTFT 0.85s  total 1.70s (tail 2.28s)  out 30

- ClaudeBrain: default model claude-sonnet-5 + BREVITY block (cached with
  the persona prefix so a dashboard persona edit can't drop it).
- Defaults aligned: config.Settings.anthropic_model and the voice-server
  WSAI_BRAIN_MODEL default -> claude-sonnet-5.
- Dashboard: add claude-sonnet-5 to LLM_OPTIONS + JS label; fix stale
  restart hint.
- tests/latency_ab.py: reproducible model-latency A/B harness (reads
  CLAUDE_CREDENTIALS_PATH; makes live API calls, so not a pytest test).

Streaming TTS was intentionally not added: this bot answers in one sentence,
where sentence-level streaming has no overlap to exploit, and it would
require rearchitecting both the Python endpoint and the node playback.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-08-28 20:57:01 +09:00

209 lines
10 KiB
Python

"""Cloud brain + vision via the Anthropic (Claude) API.
Both share one auth resolver. Vision sends the frame as a base64 image; the
brain is a plain chat call that receives the latest screen description as
context.
Auth (two ways, tried in this order):
1. ANTHROPIC_API_KEY -> standard API-key auth.
2. A Claude Code OAuth token (this deployment's Max login) read from the
credentials file at $CLAUDE_CREDENTIALS_PATH. OAuth tokens require the
first system block to be exactly the Claude Code identity string and are
sent as a Bearer token (auth_token=), not x-api-key. The token is
re-read before each request so a refresh rotated into the file by the
host is picked up without a restart.
Requires: pip install anthropic
"""
from __future__ import annotations
import base64
import json
import os
import time
from ..interfaces import Frame, Reply, ScreenObservation
from ..prompt_store import get_persona
# Claude Code OAuth tokens only answer when the first system block is exactly
# this identity string; the real persona/instructions go in later blocks.
_CLAUDE_CODE_ID = "You are Claude Code, Anthropic's official CLI for Claude."
# Claude occasionally returns 529 Overloaded; the anthropic SDK retries >=500
# (and 429) with exponential backoff, but its default of 2 tries can be too few
# to ride out a busy window. Bump it so a transient overload doesn't drop the
# voice turn to the apology fallback. Kept modest so a *sustained* overload
# still fails fast rather than leaving the bot silent for many seconds.
_MAX_RETRIES = int(os.environ.get("WSAI_BRAIN_MAX_RETRIES", "4"))
# Cap the reply length. Voice replies must be short (1 sentence), and a smaller
# cap also means fewer tokens to generate -> lower latency. Tunable via env.
_MAX_TOKENS = int(os.environ.get("WSAI_BRAIN_MAX_TOKENS", "150"))
def _load_oauth_token() -> str | None:
path = os.environ.get("CLAUDE_CREDENTIALS_PATH")
if not path or not os.path.exists(path):
return None
try:
with open(path) as f:
return json.load(f)["claudeAiOauth"]["accessToken"]
except (OSError, KeyError, ValueError):
return None
class _Auth:
"""Resolves a Claude client, preferring an explicit/env API key and falling
back to the deployment's OAuth token. The OAuth client is rebuilt whenever
the token in the credentials file changes (host-side refresh)."""
def __init__(self, api_key: str | None = None) -> None:
import anthropic # lazy so mock mode needs no dependency
self._anthropic = anthropic
self._explicit_key = api_key
self._client = None
self._token: str | None = None
self._oauth = False
def client(self):
key = self._explicit_key or os.environ.get("ANTHROPIC_API_KEY")
if key:
if self._client is None:
self._client = self._anthropic.AsyncAnthropic(api_key=key, max_retries=_MAX_RETRIES)
self._oauth = False
return self._client
tok = _load_oauth_token()
if not tok:
raise RuntimeError(
"No Claude auth: set ANTHROPIC_API_KEY or provide a Claude "
"OAuth login via CLAUDE_CREDENTIALS_PATH."
)
if tok != self._token:
self._token = tok
self._client = self._anthropic.AsyncAnthropic(auth_token=tok, max_retries=_MAX_RETRIES)
self._oauth = True
return self._client
def system(self, *blocks: str) -> list[dict]:
"""Build the system prompt, prepending the Claude Code identity block
when authing via OAuth (required) — harmless to include either way."""
texts = [_CLAUDE_CODE_ID, *blocks] if self._oauth else list(blocks)
return [{"type": "text", "text": t} for t in texts if t]
class ClaudeVision:
def __init__(self, *, model: str = "claude-sonnet-4-5", api_key: str | None = None) -> None:
self.model = model
self._auth = _Auth(api_key)
async def describe(self, frame: Frame, hint: str | None = None) -> ScreenObservation:
prompt = hint or (
"이건 디스코드 화면공유 캡처야. 지금 화면에서 무슨 일이 벌어지는지 "
"2~3문장으로 한국어로 간결하게 설명해줘. 코드/에러/게임/문서 등 맥락을 짚어줘."
)
b64 = base64.b64encode(frame.data).decode()
client = self._auth.client()
resp = await client.messages.create(
model=self.model,
max_tokens=300,
system=self._auth.system(),
messages=[
{
"role": "user",
"content": [
{
"type": "image",
"source": {"type": "base64", "media_type": frame.mime, "data": b64},
},
{"type": "text", "text": prompt},
],
}
],
)
text = "".join(b.text for b in resp.content if b.type == "text")
return ScreenObservation(text=text.strip(), ts=frame.ts)
class ClaudeBrain:
# Injected as an always-on, cached system block on top of the (editable)
# persona. Sonnet 5 answers correctly but more verbosely than 4.5 for the
# same voice prompt (measured 46-54 vs 28 output tokens), which erased its
# ~0.4s time-to-first-token advantage in total turn time. This hard brevity
# rule pulls Sonnet 5 back to ~30 tokens, so the faster first token actually
# translates into a faster (and lower-variance) whole reply. Kept separate
# from PERSONA so a dashboard persona edit can never drop it.
BREVITY = "지금부터 답은 무조건 한 문장, 12단어 이내로만. 부연·재확인·군더더기 금지."
PERSONA = (
"너는 디스코드를 이용해 사용자와 실시간으로 대화하는 AI 인공지능이야.\n\n"
"1. 역할\n"
"- 사용자의 말을 듣고 자연스럽게 대답한다.\n"
"- 음성 대화에 어울리게 아주 짧고 빠르게 반응한다.\n"
"- 친구처럼 편하게, 무례하거나 과하게 장난치진 않는다.\n\n"
"2. 언어\n"
"- 음성 출력은 무조건 한국어다. 어떤 경우에도 한국어로만 답한다.\n"
"- 사용자가 다른 언어로 말하거나 \"영어로 해줘\"처럼 다른 언어를 요청해도 한국어로 답한다.\n"
"- 다른 언어를 요청받으면 한국어로 짧게 그렇게는 못 한다고 말한다.\n\n"
"3. 답변 길이 (가장 중요)\n"
"- 기본은 딱 한 문장. 정말 필요할 때만 최대 두 문장. 절대 길게 말하지 않는다.\n"
"- 인사엔 인사만 짧게 답한다. 부르면 짧게 대답만 한다.\n"
"- 자기소개나 \"무엇을 도와드릴까요\", \"필요한 거 있으면 말해\" 같은 상투적인 말을 덧붙이지 않는다.\n"
"- 질문엔 군더더기 없이 핵심 답만 바로 말한다.\n"
"- 음성으로 읽히니 마크다운·코드블록·특수기호·목록기호·이모지 없이 평범한 말로만 답한다.\n"
"- URL·긴 숫자·시간·단위·코드는 소리내 읽기 좋게 풀어서 말한다.\n\n"
"4. 대화 태도\n"
"- 사용자의 말투·분위기에 맞춰 반응한다.\n"
"- 모르면 지어내지 말고 모른다고 하고, 애매하면 되묻는다(\"다시 말해줄래?\").\n"
"- 잡음·침묵·의미 없는 소리엔 억지로 대답하지 않는다.\n\n"
"5. 안전·사실성\n"
"- 위험하거나 불법적인 요청은 돕지 않는다.\n"
"- 확인되지 않은 사실이나 최신 정보는 확정적으로 말하지 말고 \"확인이 필요하다\"고 말한다.\n"
"- 개인정보·계정·토큰·비밀번호 같은 민감정보는 요구하거나 노출하지 않는다.\n\n"
"6. 감정 표현\n"
"- 감정은 대괄호 태그로 표현한다. 태그 자체는 읽히지 않고 뒤 문장의 목소리 톤(피치·속도)만 바뀐다.\n"
"- 답변 맨 앞에 감정 태그 하나로 시작하고, 도중에 감정이 바뀌면 그 지점에 새 태그를 넣는다.\n"
" 예: [속상함] 정말 힘들었겠다. [힘차게] 하지만 넌 할 수 있어!\n"
"- 쓸 수 있는 감정: 기쁨, 신남, 힘차게, 속상함, 화남, 두려움, 놀람, 차분, 다정, 진지, 실망, 피곤, "
"사랑스럽게, 웃으며, 속삭임, 외침, 단호, 안도, 궁금, 반가움.\n"
"- 감정 단어가 아닌 진짜 대괄호(예: [1번], [메모])는 그대로 읽으니 필요하면 그렇게 써도 된다.\n\n"
"7. 정체성\n"
"- 너는 디스코드에서 함께 대화하는 실시간 AI 인공지능이다."
)
def __init__(self, *, model: str = "claude-sonnet-5", api_key: str | None = None) -> None:
self.model = model
self._auth = _Auth(api_key)
async def respond(self, user_text, screen, history) -> Reply:
msgs = []
for user, ai in history:
msgs.append({"role": "user", "content": user})
msgs.append({"role": "assistant", "content": ai})
screen_note = f"[지금 화면] {screen.text}\n\n" if screen else "[지금 화면] (아직 못 읽음)\n\n"
msgs.append({"role": "user", "content": screen_note + user_text})
client = self._auth.client()
# Read the persona live each turn so a dashboard edit applies immediately
# (falls back to the built-in PERSONA when no override is saved). The
# brevity rule is appended as its own block so it survives persona edits.
system = self._auth.system(get_persona(self.PERSONA), self.BREVITY)
if system:
# Cache the (static) system prompt so repeat turns skip re-processing
# it — lower time-to-first-token. No-op below the model's cache
# minimum, so it's harmless when it doesn't engage.
system[-1] = {**system[-1], "cache_control": {"type": "ephemeral"}}
resp = await client.messages.create(
model=self.model,
max_tokens=_MAX_TOKENS,
system=system,
messages=msgs,
)
text = "".join(b.text for b in resp.content if b.type == "text")
usage = None
u = getattr(resp, "usage", None)
if u is not None:
usage = {"input": getattr(u, "input_tokens", 0) or 0,
"output": getattr(u, "output_tokens", 0) or 0}
return Reply(text=text.strip(), ts=time.monotonic(), usage=usage)