feat(stt/voice): better recognition, sustained barge-in, fragment discard
STT (whisper_worker): add VAD tuning (speech_pad_ms so soft first/last words
aren't clipped), condition_on_previous_text=False + temperature fallback +
no_speech/logprob/compression thresholds to reject noisy/quiet decodes, drop
per-segment non-speech, and a hallucination guard that blanks Whisper's classic
Korean silence/noise boilerplate ("감사합니다" 등) when no_speech_prob is high.
Barge-in (bot.mjs): stop the bot's TTS only when the speaking (green ring) stays
on for >= WSAI_BARGE_IN_MS (default 700ms), not on the first blip — cancelled if
speaking stops in time. AfterSilence 800->1000ms so trailing soft words finish.
Fragment gate (dashboard): a lone connective filler ("그러면") carries no
answerable intent -> discard as [대기] instead of replying; hidden by the noise
filter like [잡음].
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
@@ -23,6 +23,31 @@ import os
|
||||
import sys
|
||||
import time
|
||||
|
||||
# Whisper's classic Korean hallucinations on silence/noise/keyboard clatter —
|
||||
# it "hears" video-outro boilerplate. Drop these when the whole utterance is one
|
||||
# of them AND the segment looked like non-speech, so a real "감사합니다" survives.
|
||||
_HALLUCINATIONS = {
|
||||
"감사합니다", "고맙습니다", "감사합니다.", "고맙습니다.",
|
||||
"시청해주셔서 감사합니다", "시청해 주셔서 감사합니다", "끝까지 시청해주셔서 감사합니다",
|
||||
"구독과 좋아요 부탁드립니다", "구독 좋아요 부탁드립니다", "다음 영상에서 만나요",
|
||||
"다음 시간에 만나요", "안녕히 계세요",
|
||||
}
|
||||
|
||||
|
||||
def _norm(t: str) -> str:
|
||||
return t.strip().rstrip(" .!?…~").strip()
|
||||
|
||||
|
||||
def _guard_hallucination(text: str, worst_no_speech: float) -> str:
|
||||
"""Blank out a lone known-hallucination phrase when the audio was probably
|
||||
not speech (high no_speech_prob)."""
|
||||
n = _norm(text)
|
||||
if not n:
|
||||
return ""
|
||||
if n in {_norm(h) for h in _HALLUCINATIONS} and worst_no_speech > 0.5:
|
||||
return ""
|
||||
return text
|
||||
|
||||
# Split protocol from library noise BEFORE importing anything heavy.
|
||||
_proto = os.fdopen(os.dup(1), "w", buffering=1) # private copy of real stdout
|
||||
os.dup2(2, 1) # fd1 -> stderr, so stray library prints don't hit the protocol
|
||||
@@ -108,9 +133,31 @@ def main() -> None:
|
||||
wav,
|
||||
language=language,
|
||||
beam_size=int(req.get("beam_size", 5)),
|
||||
# VAD strips non-speech (keyboard clatter, room noise, silence)
|
||||
# before decoding, which both improves accuracy and kills most
|
||||
# hallucinations. speech_pad_ms keeps a little lead/trail so soft
|
||||
# first/last words aren't clipped.
|
||||
vad_filter=bool(req.get("vad_filter", True)),
|
||||
vad_parameters=dict(min_silence_duration_ms=300, speech_pad_ms=250),
|
||||
# Don't feed the previous text back in — that's what makes Whisper
|
||||
# loop/hallucinate. Temperature fallback + thresholds reject
|
||||
# low-confidence (noisy/quiet) decodes instead of inventing words.
|
||||
condition_on_previous_text=False,
|
||||
temperature=[0.0, 0.2, 0.4, 0.6],
|
||||
no_speech_threshold=0.6,
|
||||
log_prob_threshold=-1.0,
|
||||
compression_ratio_threshold=2.4,
|
||||
)
|
||||
text = "".join(seg.text for seg in segments).strip()
|
||||
parts, worst_ns = [], 0.0
|
||||
for seg in segments:
|
||||
nsp = float(getattr(seg, "no_speech_prob", 0.0) or 0.0)
|
||||
alp = float(getattr(seg, "avg_logprob", 0.0) or 0.0)
|
||||
# Drop a segment that is almost certainly non-speech noise.
|
||||
if nsp > 0.8 and alp < -0.4:
|
||||
continue
|
||||
worst_ns = max(worst_ns, nsp)
|
||||
parts.append(seg.text)
|
||||
text = _guard_hallucination("".join(parts).strip(), worst_ns)
|
||||
ms = int((time.monotonic() - s) * 1000)
|
||||
_emit({"ok": True, "text": text, "language": info.language, "ms": ms})
|
||||
except Exception as exc: # keep the worker alive across bad requests
|
||||
|
||||
Reference in New Issue
Block a user