STT (whisper_worker): add VAD tuning (speech_pad_ms so soft first/last words
aren't clipped), condition_on_previous_text=False + temperature fallback +
no_speech/logprob/compression thresholds to reject noisy/quiet decodes, drop
per-segment non-speech, and a hallucination guard that blanks Whisper's classic
Korean silence/noise boilerplate ("감사합니다" 등) when no_speech_prob is high.
Barge-in (bot.mjs): stop the bot's TTS only when the speaking (green ring) stays
on for >= WSAI_BARGE_IN_MS (default 700ms), not on the first blip — cancelled if
speaking stops in time. AfterSilence 800->1000ms so trailing soft words finish.
Fragment gate (dashboard): a lone connective filler ("그러면") carries no
answerable intent -> discard as [대기] instead of replying; hidden by the noise
filter like [잡음].
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
170 lines
6.9 KiB
Python
170 lines
6.9 KiB
Python
"""Persistent faster-whisper STT worker.
|
|
|
|
faster-whisper (ctranslate2) lives in its own Python (whisper312); loading the
|
|
model takes seconds, so we load it ONCE here and then serve transcription
|
|
requests over stdin/stdout. This process is launched with the whisper312
|
|
interpreter by wsai.backends.whisper.WhisperSTT.
|
|
|
|
Like the MeloTTS worker, model/backend chatter could corrupt the JSON protocol,
|
|
so on startup we split the streams: a private duplicate of the original stdout
|
|
carries the protocol, and fd 1 is redirected to fd 2 so any library print lands
|
|
on stderr instead (where the parent drains it for diagnostics).
|
|
|
|
Protocol (one JSON object per line, on the protocol channel):
|
|
<- {"wav": "/abs/path.wav", "language": "ko"}
|
|
-> {"ok": true, "text": "...", "language": "ko", "ms": 123}
|
|
-> {"ok": false, "error": "..."}
|
|
On startup, once the model is ready, it emits exactly one line:
|
|
-> {"ready": true, "ms": <load-ms>, "device": "cpu", "model": "small"}
|
|
"""
|
|
|
|
import json
|
|
import os
|
|
import sys
|
|
import time
|
|
|
|
# Whisper's classic Korean hallucinations on silence/noise/keyboard clatter —
|
|
# it "hears" video-outro boilerplate. Drop these when the whole utterance is one
|
|
# of them AND the segment looked like non-speech, so a real "감사합니다" survives.
|
|
_HALLUCINATIONS = {
|
|
"감사합니다", "고맙습니다", "감사합니다.", "고맙습니다.",
|
|
"시청해주셔서 감사합니다", "시청해 주셔서 감사합니다", "끝까지 시청해주셔서 감사합니다",
|
|
"구독과 좋아요 부탁드립니다", "구독 좋아요 부탁드립니다", "다음 영상에서 만나요",
|
|
"다음 시간에 만나요", "안녕히 계세요",
|
|
}
|
|
|
|
|
|
def _norm(t: str) -> str:
|
|
return t.strip().rstrip(" .!?…~").strip()
|
|
|
|
|
|
def _guard_hallucination(text: str, worst_no_speech: float) -> str:
|
|
"""Blank out a lone known-hallucination phrase when the audio was probably
|
|
not speech (high no_speech_prob)."""
|
|
n = _norm(text)
|
|
if not n:
|
|
return ""
|
|
if n in {_norm(h) for h in _HALLUCINATIONS} and worst_no_speech > 0.5:
|
|
return ""
|
|
return text
|
|
|
|
# Split protocol from library noise BEFORE importing anything heavy.
|
|
_proto = os.fdopen(os.dup(1), "w", buffering=1) # private copy of real stdout
|
|
os.dup2(2, 1) # fd1 -> stderr, so stray library prints don't hit the protocol
|
|
|
|
|
|
def _emit(obj: dict) -> None:
|
|
_proto.write(json.dumps(obj, ensure_ascii=False) + "\n")
|
|
_proto.flush()
|
|
|
|
|
|
def _log(*a):
|
|
print(*a, file=sys.stderr, flush=True)
|
|
|
|
|
|
def main() -> None:
|
|
model_name = os.environ.get("WSAI_WHISPER_MODEL", "small")
|
|
requested = os.environ.get("WSAI_WHISPER_DEVICE", "auto") # cpu | cuda | auto
|
|
default_lang = os.environ.get("WSAI_WHISPER_LANGUAGE", "ko") or None
|
|
|
|
from faster_whisper import WhisperModel # heavy import; only in whisper venv
|
|
|
|
def _has_cuda() -> bool:
|
|
try:
|
|
import ctranslate2
|
|
|
|
return ctranslate2.get_cuda_device_count() > 0
|
|
except Exception:
|
|
return False
|
|
|
|
device = requested
|
|
if requested == "auto":
|
|
device = "cuda" if _has_cuda() else "cpu"
|
|
|
|
def _compute_for(dev: str) -> str:
|
|
# int8 on CPU keeps a small model fast; float16 is the usual CUDA choice.
|
|
return os.environ.get(
|
|
"WSAI_WHISPER_COMPUTE", "int8" if dev == "cpu" else "float16"
|
|
)
|
|
|
|
t0 = time.monotonic()
|
|
try:
|
|
model = WhisperModel(model_name, device=device, compute_type=_compute_for(device))
|
|
except Exception as exc:
|
|
# CUDA picked but unusable (missing libs, OOM): fall back to CPU rather
|
|
# than leaving the whole voice loop dead.
|
|
if device == "cuda":
|
|
_log(f"[whisper_worker] CUDA load failed ({exc}); falling back to CPU")
|
|
device = "cpu"
|
|
model = WhisperModel(model_name, device=device, compute_type=_compute_for(device))
|
|
else:
|
|
raise
|
|
compute = _compute_for(device)
|
|
load_ms = int((time.monotonic() - t0) * 1000)
|
|
|
|
# Warm up before signalling ready: the first CUDA transcribe pays a large
|
|
# lazy cost (kernel autotune), which would slow the first real utterance.
|
|
# Run a dummy transcribe on 1s of silence here so "ready" means "hot".
|
|
warmup_ms = None
|
|
try:
|
|
import numpy as np
|
|
|
|
w = time.monotonic()
|
|
segs, _ = model.transcribe(np.zeros(16000, dtype=np.float32), language=default_lang)
|
|
for _ in segs: # segments are lazy; drain to force the actual compute
|
|
pass
|
|
warmup_ms = int((time.monotonic() - w) * 1000)
|
|
except Exception as exc:
|
|
_log(f"[whisper_worker] warmup skipped: {exc}")
|
|
|
|
_emit({"ready": True, "ms": load_ms, "device": device, "model": model_name, "warmup_ms": warmup_ms})
|
|
_log(f"[whisper_worker] {model_name} ready in {load_ms} ms on {device}/{compute} (warmup {warmup_ms} ms)")
|
|
|
|
for line in sys.stdin:
|
|
line = line.strip()
|
|
if not line:
|
|
continue
|
|
try:
|
|
req = json.loads(line)
|
|
wav = req["wav"]
|
|
language = req.get("language", default_lang)
|
|
s = time.monotonic()
|
|
segments, info = model.transcribe(
|
|
wav,
|
|
language=language,
|
|
beam_size=int(req.get("beam_size", 5)),
|
|
# VAD strips non-speech (keyboard clatter, room noise, silence)
|
|
# before decoding, which both improves accuracy and kills most
|
|
# hallucinations. speech_pad_ms keeps a little lead/trail so soft
|
|
# first/last words aren't clipped.
|
|
vad_filter=bool(req.get("vad_filter", True)),
|
|
vad_parameters=dict(min_silence_duration_ms=300, speech_pad_ms=250),
|
|
# Don't feed the previous text back in — that's what makes Whisper
|
|
# loop/hallucinate. Temperature fallback + thresholds reject
|
|
# low-confidence (noisy/quiet) decodes instead of inventing words.
|
|
condition_on_previous_text=False,
|
|
temperature=[0.0, 0.2, 0.4, 0.6],
|
|
no_speech_threshold=0.6,
|
|
log_prob_threshold=-1.0,
|
|
compression_ratio_threshold=2.4,
|
|
)
|
|
parts, worst_ns = [], 0.0
|
|
for seg in segments:
|
|
nsp = float(getattr(seg, "no_speech_prob", 0.0) or 0.0)
|
|
alp = float(getattr(seg, "avg_logprob", 0.0) or 0.0)
|
|
# Drop a segment that is almost certainly non-speech noise.
|
|
if nsp > 0.8 and alp < -0.4:
|
|
continue
|
|
worst_ns = max(worst_ns, nsp)
|
|
parts.append(seg.text)
|
|
text = _guard_hallucination("".join(parts).strip(), worst_ns)
|
|
ms = int((time.monotonic() - s) * 1000)
|
|
_emit({"ok": True, "text": text, "language": info.language, "ms": ms})
|
|
except Exception as exc: # keep the worker alive across bad requests
|
|
_emit({"ok": False, "error": f"{type(exc).__name__}: {exc}"})
|
|
_log(f"[whisper_worker] error: {exc}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|