Sonnet 5 reaches first token ~0.4s sooner than Sonnet 4.5 but answers the same voice prompt more verbosely (measured 46-54 vs ~28 output tokens), which erased the win in total turn time. Add a cached, persona-independent brevity system block that pulls output back to ~30 tokens, so the faster first token becomes a faster, lower-variance whole reply. Measured (16-round interleaved A/B, production-shaped call): sonnet-4-5 TTFT 1.26s total 2.02s (tail 3.26s) out 29 sonnet-5+brev TTFT 0.85s total 1.70s (tail 2.28s) out 30 - ClaudeBrain: default model claude-sonnet-5 + BREVITY block (cached with the persona prefix so a dashboard persona edit can't drop it). - Defaults aligned: config.Settings.anthropic_model and the voice-server WSAI_BRAIN_MODEL default -> claude-sonnet-5. - Dashboard: add claude-sonnet-5 to LLM_OPTIONS + JS label; fix stale restart hint. - tests/latency_ab.py: reproducible model-latency A/B harness (reads CLAUDE_CREDENTIALS_PATH; makes live API calls, so not a pytest test). Streaming TTS was intentionally not added: this bot answers in one sentence, where sentence-level streaming has no overlap to exploit, and it would require rearchitecting both the Python endpoint and the node playback. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
63 lines
2.2 KiB
Python
63 lines
2.2 KiB
Python
"""Configuration. Each field names a backend; the factory maps names -> classes.
|
|
|
|
Defaults are all "mock" so the skeleton runs out of the box. Flip individual
|
|
fields (via env or code) as real backends land.
|
|
|
|
Env overrides (optional):
|
|
WSAI_SOURCE, WSAI_VISION, WSAI_STT, WSAI_TTS, WSAI_BRAIN, WSAI_TEXT
|
|
WSAI_CAPTURE_INTERVAL
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
from dataclasses import dataclass
|
|
|
|
|
|
@dataclass
|
|
class Settings:
|
|
source: str | None = "mock" # mock | mss | None (eyes-free)
|
|
vision: str | None = "mock" # mock | claude | None (eyes-free)
|
|
stt: str | None = "mock" # mock | whisper | None
|
|
tts: str | None = "mock" # mock | melo | None
|
|
brain: str = "mock" # mock | claude
|
|
text: str | None = None # None | (discord)
|
|
|
|
capture_interval: float = 1.5
|
|
anthropic_model: str = "claude-sonnet-5"
|
|
|
|
@classmethod
|
|
def from_env(cls) -> "Settings":
|
|
def opt(name: str, default):
|
|
v = os.environ.get(name)
|
|
return default if v is None else (None if v.lower() == "none" else v)
|
|
|
|
return cls(
|
|
source=opt("WSAI_SOURCE", "mock"),
|
|
vision=opt("WSAI_VISION", "mock"),
|
|
stt=opt("WSAI_STT", "mock"),
|
|
tts=opt("WSAI_TTS", "mock"),
|
|
brain=opt("WSAI_BRAIN", "mock"),
|
|
text=opt("WSAI_TEXT", None),
|
|
capture_interval=float(os.environ.get("WSAI_CAPTURE_INTERVAL", "1.5")),
|
|
)
|
|
|
|
@classmethod
|
|
def mock(cls) -> "Settings":
|
|
return cls()
|
|
|
|
@classmethod
|
|
def live(cls) -> "Settings":
|
|
"""A realistic local config: capture this screen, Claude eyes+brain,
|
|
mock voice (until STT/TTS backends are wired)."""
|
|
return cls(source="mss", vision="claude", brain="claude", stt="mock", tts="mock")
|
|
|
|
@classmethod
|
|
def voice(cls) -> "Settings":
|
|
"""Eyes-free voice loop: no screen share, just STT -> Brain -> TTS.
|
|
|
|
Screen capture is deferred, so source/vision are off. Backends default
|
|
to mock so it runs out of the box; flip stt/tts/brain to real ones as
|
|
they land."""
|
|
return cls(source=None, vision=None, stt="mock", tts="mock", brain="mock")
|