feat(tts): port glyph-speed/word-gap/sentence-gap/pitch controls from tts_site

Ports the 4 independent voice controls documented in tts_site's manual into the
MeloTTS backend, layered on top of the existing emotion-tag segments:

- glyph speed: generation-stage length_scale (existing speed path); default
  bumped 1.2 -> 1.25 to match the manual (still well below the 1.5 that slurred)
- word_gap (sec, -0.2..0.5): scale intra-sentence silences after natural synth
- sentence_gap (sec, -0.5..1.5): insert/trim silence at sentence boundaries,
  replacing the old fixed 120ms inter-segment gap
- pitch (semitones, -12..12): global offset added on top of per-emotion pitch

Each reply is split into sentences, synthesised per-sentence at its segment's
speed, word_gap applied, joined with sentence_gap, then pitch-shifted. Exposed
via WSAI_TTS_SPEED/WORD_GAP/SENTENCE_GAP/PITCH env vars and MeloTTS ctor args.

Verified by real synthesis: each control independently changes wav duration
(speed, sentence_gap, word_gap-with-pauses, pitch); 29 tests pass.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
EJClaw
2026-08-26 22:06:21 +09:00
parent a207ae05c5
commit 0b92284ff8
2 changed files with 164 additions and 24 deletions

View File

@@ -12,7 +12,10 @@ Env:
WSAI_MELO_DEVICE cpu | cuda | auto (default auto: GPU if torch sees one,
else CPU; the worker falls back to CPU if CUDA fails)
WSAI_TTS_OUT_DIR where wavs are written (default ~/.cache/wsai/tts)
WSAI_TTS_SPEED synthesis speed multiplier (default 1.2)
WSAI_TTS_SPEED glyph speed / length_scale multiplier (0.5..2.0, default 1.25)
WSAI_TTS_WORD_GAP intra-sentence pause delta, sec (-0.2..0.5, default 0.25)
WSAI_TTS_SENTENCE_GAP sentence-boundary silence, sec (-0.5..1.5, default 0.75)
WSAI_TTS_PITCH global semitone offset (-12..12, default 0.0)
"""
from __future__ import annotations
@@ -86,6 +89,9 @@ class MeloTTS:
device: str | None = None,
out_dir: str | None = None,
speed: float | None = None,
word_gap: float | None = None,
sentence_gap: float | None = None,
pitch: float | None = None,
sink: Sink | None = None,
) -> None:
self.python = python or os.environ.get("WSAI_MELO_PYTHON", _DEFAULT_PYTHON)
@@ -93,7 +99,14 @@ class MeloTTS:
self.out_dir = Path(out_dir or os.environ.get("WSAI_TTS_OUT_DIR")
or (Path.home() / ".cache/wsai/tts"))
self.speed = float(speed if speed is not None
else os.environ.get("WSAI_TTS_SPEED", "1.2"))
else os.environ.get("WSAI_TTS_SPEED", "1.25"))
# Reply-global rhythm/pitch controls (see melo_worker.py / docs manual).
self.word_gap = float(word_gap if word_gap is not None
else os.environ.get("WSAI_TTS_WORD_GAP", "0.25"))
self.sentence_gap = float(sentence_gap if sentence_gap is not None
else os.environ.get("WSAI_TTS_SENTENCE_GAP", "0.75"))
self.pitch = float(pitch if pitch is not None
else os.environ.get("WSAI_TTS_PITCH", "0.0"))
self.sink = sink or _log_sink
self._proc: asyncio.subprocess.Process | None = None
self._lock = asyncio.Lock()
@@ -202,9 +215,19 @@ class MeloTTS:
for s in segments
],
"out": out,
"word_gap": self.word_gap,
"sentence_gap": self.sentence_gap,
"pitch": self.pitch,
}
else: # empty/whitespace reply: keep legacy single-utterance behaviour
payload = {"text": text, "out": out, "speed": self.speed}
payload = {
"text": text,
"out": out,
"speed": self.speed,
"word_gap": self.word_gap,
"sentence_gap": self.sentence_gap,
"pitch": self.pitch,
}
req = json.dumps(payload)
s = time.monotonic()
async with self._lock: