diff --git a/wsai/backends/claude.py b/wsai/backends/claude.py index 7225371..e9fe258 100644 --- a/wsai/backends/claude.py +++ b/wsai/backends/claude.py @@ -37,6 +37,10 @@ _CLAUDE_CODE_ID = "You are Claude Code, Anthropic's official CLI for Claude." # still fails fast rather than leaving the bot silent for many seconds. _MAX_RETRIES = int(os.environ.get("WSAI_BRAIN_MAX_RETRIES", "4")) +# Cap the reply length. Voice replies must be short (1 sentence), and a smaller +# cap also means fewer tokens to generate -> lower latency. Tunable via env. +_MAX_TOKENS = int(os.environ.get("WSAI_BRAIN_MAX_TOKENS", "150")) + def _load_oauth_token() -> str | None: path = os.environ.get("CLAUDE_CREDENTIALS_PATH") @@ -127,15 +131,18 @@ class ClaudeBrain: "너는 디스코드를 이용해 사용자와 실시간으로 대화하는 AI 인공지능이야.\n\n" "1. 역할\n" "- 사용자의 말을 듣고 자연스럽게 대답한다.\n" - "- 음성 대화에 어울리게 짧고 빠르게 반응한다.\n" + "- 음성 대화에 어울리게 아주 짧고 빠르게 반응한다.\n" "- 친구처럼 편하게, 무례하거나 과하게 장난치진 않는다.\n\n" "2. 언어\n" "- 음성 출력은 무조건 한국어다. 어떤 경우에도 한국어로만 답한다.\n" "- 사용자가 다른 언어로 말하거나 \"영어로 해줘\"처럼 다른 언어를 요청해도 한국어로 답한다.\n" "- 다른 언어를 요청받으면 한국어로 짧게 그렇게는 못 한다고 말한다.\n\n" - "3. 답변 방식\n" + "3. 답변 길이 (가장 중요)\n" + "- 기본은 딱 한 문장. 정말 필요할 때만 최대 두 문장. 절대 길게 말하지 않는다.\n" + "- 인사엔 인사만 짧게 답한다. 부르면 짧게 대답만 한다.\n" + "- 자기소개나 \"무엇을 도와드릴까요\", \"필요한 거 있으면 말해\" 같은 상투적인 말을 덧붙이지 않는다.\n" + "- 질문엔 군더더기 없이 핵심 답만 바로 말한다.\n" "- 음성으로 읽히니 마크다운·코드블록·특수기호·목록기호·이모지 없이 평범한 말로만 답한다.\n" - "- 기본은 한두 문장, 길어도 10초 안팎. 길어질 땐 핵심부터 말하고 필요하면 이어서 설명한다.\n" "- URL·긴 숫자·시간·단위·코드는 소리내 읽기 좋게 풀어서 말한다.\n\n" "4. 대화 태도\n" "- 사용자의 말투·분위기에 맞춰 반응한다.\n" @@ -170,10 +177,16 @@ class ClaudeBrain: client = self._auth.client() # Read the persona live each turn so a dashboard edit applies immediately # (falls back to the built-in PERSONA when no override is saved). + system = self._auth.system(get_persona(self.PERSONA)) + if system: + # Cache the (static) system prompt so repeat turns skip re-processing + # it — lower time-to-first-token. No-op below the model's cache + # minimum, so it's harmless when it doesn't engage. + system[-1] = {**system[-1], "cache_control": {"type": "ephemeral"}} resp = await client.messages.create( model=self.model, - max_tokens=400, - system=self._auth.system(get_persona(self.PERSONA)), + max_tokens=_MAX_TOKENS, + system=system, messages=msgs, ) text = "".join(b.text for b in resp.content if b.type == "text")