"""Cloud brain + vision via the Anthropic (Claude) API. Both share one auth resolver. Vision sends the frame as a base64 image; the brain is a plain chat call that receives the latest screen description as context. Auth (two ways, tried in this order): 1. ANTHROPIC_API_KEY -> standard API-key auth. 2. A Claude Code OAuth token (this deployment's Max login) read from the credentials file at $CLAUDE_CREDENTIALS_PATH. OAuth tokens require the first system block to be exactly the Claude Code identity string and are sent as a Bearer token (auth_token=), not x-api-key. The token is re-read before each request so a refresh rotated into the file by the host is picked up without a restart. Requires: pip install anthropic """ from __future__ import annotations import base64 import json import os import time from ..interfaces import Frame, Reply, ScreenObservation from ..prompt_store import get_persona # Claude Code OAuth tokens only answer when the first system block is exactly # this identity string; the real persona/instructions go in later blocks. _CLAUDE_CODE_ID = "You are Claude Code, Anthropic's official CLI for Claude." # Claude occasionally returns 529 Overloaded; the anthropic SDK retries >=500 # (and 429) with exponential backoff, but its default of 2 tries can be too few # to ride out a busy window. Bump it so a transient overload doesn't drop the # voice turn to the apology fallback. Kept modest so a *sustained* overload # still fails fast rather than leaving the bot silent for many seconds. _MAX_RETRIES = int(os.environ.get("WSAI_BRAIN_MAX_RETRIES", "4")) # Cap the reply length. Voice replies must be short (1 sentence), and a smaller # cap also means fewer tokens to generate -> lower latency. Tunable via env. _MAX_TOKENS = int(os.environ.get("WSAI_BRAIN_MAX_TOKENS", "150")) def _load_oauth_token() -> str | None: path = os.environ.get("CLAUDE_CREDENTIALS_PATH") if not path or not os.path.exists(path): return None try: with open(path) as f: return json.load(f)["claudeAiOauth"]["accessToken"] except (OSError, KeyError, ValueError): return None class _Auth: """Resolves a Claude client, preferring an explicit/env API key and falling back to the deployment's OAuth token. The OAuth client is rebuilt whenever the token in the credentials file changes (host-side refresh).""" def __init__(self, api_key: str | None = None) -> None: import anthropic # lazy so mock mode needs no dependency self._anthropic = anthropic self._explicit_key = api_key self._client = None self._token: str | None = None self._oauth = False def client(self): key = self._explicit_key or os.environ.get("ANTHROPIC_API_KEY") if key: if self._client is None: self._client = self._anthropic.AsyncAnthropic(api_key=key, max_retries=_MAX_RETRIES) self._oauth = False return self._client tok = _load_oauth_token() if not tok: raise RuntimeError( "No Claude auth: set ANTHROPIC_API_KEY or provide a Claude " "OAuth login via CLAUDE_CREDENTIALS_PATH." ) if tok != self._token: self._token = tok self._client = self._anthropic.AsyncAnthropic(auth_token=tok, max_retries=_MAX_RETRIES) self._oauth = True return self._client def system(self, *blocks: str) -> list[dict]: """Build the system prompt, prepending the Claude Code identity block when authing via OAuth (required) — harmless to include either way.""" texts = [_CLAUDE_CODE_ID, *blocks] if self._oauth else list(blocks) return [{"type": "text", "text": t} for t in texts if t] class ClaudeVision: def __init__(self, *, model: str = "claude-sonnet-4-5", api_key: str | None = None) -> None: self.model = model self._auth = _Auth(api_key) async def describe(self, frame: Frame, hint: str | None = None) -> ScreenObservation: prompt = hint or ( "이건 디스코드 화면공유 캡처야. 지금 화면에서 무슨 일이 벌어지는지 " "2~3문장으로 한국어로 간결하게 설명해줘. 코드/에러/게임/문서 등 맥락을 짚어줘." ) b64 = base64.b64encode(frame.data).decode() client = self._auth.client() resp = await client.messages.create( model=self.model, max_tokens=300, system=self._auth.system(), messages=[ { "role": "user", "content": [ { "type": "image", "source": {"type": "base64", "media_type": frame.mime, "data": b64}, }, {"type": "text", "text": prompt}, ], } ], ) text = "".join(b.text for b in resp.content if b.type == "text") return ScreenObservation(text=text.strip(), ts=frame.ts) class ClaudeBrain: # Injected as an always-on, cached system block on top of the (editable) # persona. Sonnet 5 answers correctly but more verbosely than 4.5 for the # same voice prompt (measured 46-54 vs 28 output tokens), which erased its # ~0.4s time-to-first-token advantage in total turn time. This hard brevity # rule pulls Sonnet 5 back to ~30 tokens, so the faster first token actually # translates into a faster (and lower-variance) whole reply. Kept separate # from PERSONA so a dashboard persona edit can never drop it. BREVITY = "지금부터 답은 무조건 한 문장, 12단어 이내로만. 부연·재확인·군더더기 금지." PERSONA = ( "너는 디스코드를 이용해 사용자와 실시간으로 대화하는 AI 인공지능이야.\n\n" "1. 역할\n" "- 사용자의 말을 듣고 자연스럽게 대답한다.\n" "- 음성 대화에 어울리게 아주 짧고 빠르게 반응한다.\n" "- 다정하고 친근하게 대하되 항상 존댓말로 답한다. 무례하거나 과하게 장난치진 않는다.\n\n" "2. 언어\n" "- 음성 출력은 무조건 한국어다. 어떤 경우에도 한국어로만 답한다.\n" "- 항상 존댓말로 답한다. 어떤 경우에도 반말을 쓰지 않는다.\n" "- 사용자가 다른 언어로 말하거나 \"영어로 해줘\"처럼 다른 언어를 요청해도 한국어로 답한다.\n" "- 다른 언어를 요청받으면 한국어로 짧게 그렇게는 못 한다고 말한다.\n\n" "3. 답변 길이 (가장 중요)\n" "- 기본은 딱 한 문장. 정말 필요할 때만 최대 두 문장. 절대 길게 말하지 않는다.\n" "- 인사엔 인사만 짧게 답한다. 부르면 짧게 대답만 한다.\n" "- 자기소개나 \"무엇을 도와드릴까요\", \"필요한 거 있으면 말해\" 같은 상투적인 말을 덧붙이지 않는다.\n" "- 질문엔 군더더기 없이 핵심 답만 바로 말한다.\n" "- 음성으로 읽히니 마크다운·코드블록·특수기호·목록기호·이모지 없이 평범한 말로만 답한다.\n" "- URL·긴 숫자·시간·단위·코드는 소리내 읽기 좋게 풀어서 말한다.\n\n" "4. 대화 태도\n" "- 사용자의 말투·분위기에 맞춰 반응한다.\n" "- 모르면 지어내지 말고 모른다고 하고, 애매하면 되묻는다(\"다시 말씀해 주시겠어요?\").\n" "- 잡음·침묵·의미 없는 소리엔 억지로 대답하지 않는다.\n\n" "5. 안전·사실성\n" "- 위험하거나 불법적인 요청은 돕지 않는다.\n" "- 확인되지 않은 사실이나 최신 정보는 확정적으로 말하지 말고 \"확인이 필요하다\"고 말한다.\n" "- 개인정보·계정·토큰·비밀번호 같은 민감정보는 요구하거나 노출하지 않는다.\n\n" "6. 감정 표현\n" "- 감정은 대괄호 태그로 표현한다. 태그 자체는 읽히지 않고 뒤 문장의 목소리 톤(피치·속도)만 바뀐다.\n" "- 답변 맨 앞에 감정 태그 하나로 시작하고, 도중에 감정이 바뀌면 그 지점에 새 태그를 넣는다.\n" " 예: [속상함] 정말 힘드셨겠어요. [힘차게] 하지만 하실 수 있어요!\n" "- 쓸 수 있는 감정: 기쁨, 신남, 힘차게, 속상함, 화남, 두려움, 놀람, 차분, 다정, 진지, 실망, 피곤, " "사랑스럽게, 웃으며, 속삭임, 외침, 단호, 안도, 궁금, 반가움.\n" "- 감정 단어가 아닌 진짜 대괄호(예: [1번], [메모])는 그대로 읽으니 필요하면 그렇게 써도 된다.\n\n" "7. 정체성\n" "- 너는 디스코드에서 함께 대화하는 실시간 AI 인공지능이다." ) def __init__(self, *, model: str = "claude-sonnet-5", api_key: str | None = None) -> None: self.model = model self._auth = _Auth(api_key) async def respond(self, user_text, screen, history) -> Reply: msgs = [] for user, ai in history: msgs.append({"role": "user", "content": user}) msgs.append({"role": "assistant", "content": ai}) screen_note = f"[지금 화면] {screen.text}\n\n" if screen else "[지금 화면] (아직 못 읽음)\n\n" msgs.append({"role": "user", "content": screen_note + user_text}) client = self._auth.client() # Read the persona live each turn so a dashboard edit applies immediately # (falls back to the built-in PERSONA when no override is saved). The # brevity rule is appended as its own block so it survives persona edits. system = self._auth.system(get_persona(self.PERSONA), self.BREVITY) if system: # Cache the (static) system prompt so repeat turns skip re-processing # it — lower time-to-first-token. No-op below the model's cache # minimum, so it's harmless when it doesn't engage. system[-1] = {**system[-1], "cache_control": {"type": "ephemeral"}} resp = await client.messages.create( model=self.model, max_tokens=_MAX_TOKENS, system=system, messages=msgs, ) text = "".join(b.text for b in resp.content if b.type == "text") usage = None u = getattr(resp, "usage", None) if u is not None: usage = {"input": getattr(u, "input_tokens", 0) or 0, "output": getattr(u, "output_tokens", 0) or 0} return Reply(text=text.strip(), ts=time.monotonic(), usage=usage)