From 0b92284ff89b4db0640a9689fb720118f77e7be8 Mon Sep 17 00:00:00 2001 From: EJClaw Date: Wed, 26 Aug 2026 22:06:21 +0900 Subject: [PATCH] feat(tts): port glyph-speed/word-gap/sentence-gap/pitch controls from tts_site Ports the 4 independent voice controls documented in tts_site's manual into the MeloTTS backend, layered on top of the existing emotion-tag segments: - glyph speed: generation-stage length_scale (existing speed path); default bumped 1.2 -> 1.25 to match the manual (still well below the 1.5 that slurred) - word_gap (sec, -0.2..0.5): scale intra-sentence silences after natural synth - sentence_gap (sec, -0.5..1.5): insert/trim silence at sentence boundaries, replacing the old fixed 120ms inter-segment gap - pitch (semitones, -12..12): global offset added on top of per-emotion pitch Each reply is split into sentences, synthesised per-sentence at its segment's speed, word_gap applied, joined with sentence_gap, then pitch-shifted. Exposed via WSAI_TTS_SPEED/WORD_GAP/SENTENCE_GAP/PITCH env vars and MeloTTS ctor args. Verified by real synthesis: each control independently changes wav duration (speed, sentence_gap, word_gap-with-pauses, pitch); 29 tests pass. Co-Authored-By: Claude Opus 4.7 --- wsai/backends/melo.py | 29 ++++++- wsai/backends/melo_worker.py | 159 ++++++++++++++++++++++++++++++----- 2 files changed, 164 insertions(+), 24 deletions(-) diff --git a/wsai/backends/melo.py b/wsai/backends/melo.py index 3e15bac..54caf38 100644 --- a/wsai/backends/melo.py +++ b/wsai/backends/melo.py @@ -12,7 +12,10 @@ Env: WSAI_MELO_DEVICE cpu | cuda | auto (default auto: GPU if torch sees one, else CPU; the worker falls back to CPU if CUDA fails) WSAI_TTS_OUT_DIR where wavs are written (default ~/.cache/wsai/tts) - WSAI_TTS_SPEED synthesis speed multiplier (default 1.2) + WSAI_TTS_SPEED glyph speed / length_scale multiplier (0.5..2.0, default 1.25) + WSAI_TTS_WORD_GAP intra-sentence pause delta, sec (-0.2..0.5, default 0.25) + WSAI_TTS_SENTENCE_GAP sentence-boundary silence, sec (-0.5..1.5, default 0.75) + WSAI_TTS_PITCH global semitone offset (-12..12, default 0.0) """ from __future__ import annotations @@ -86,6 +89,9 @@ class MeloTTS: device: str | None = None, out_dir: str | None = None, speed: float | None = None, + word_gap: float | None = None, + sentence_gap: float | None = None, + pitch: float | None = None, sink: Sink | None = None, ) -> None: self.python = python or os.environ.get("WSAI_MELO_PYTHON", _DEFAULT_PYTHON) @@ -93,7 +99,14 @@ class MeloTTS: self.out_dir = Path(out_dir or os.environ.get("WSAI_TTS_OUT_DIR") or (Path.home() / ".cache/wsai/tts")) self.speed = float(speed if speed is not None - else os.environ.get("WSAI_TTS_SPEED", "1.2")) + else os.environ.get("WSAI_TTS_SPEED", "1.25")) + # Reply-global rhythm/pitch controls (see melo_worker.py / docs manual). + self.word_gap = float(word_gap if word_gap is not None + else os.environ.get("WSAI_TTS_WORD_GAP", "0.25")) + self.sentence_gap = float(sentence_gap if sentence_gap is not None + else os.environ.get("WSAI_TTS_SENTENCE_GAP", "0.75")) + self.pitch = float(pitch if pitch is not None + else os.environ.get("WSAI_TTS_PITCH", "0.0")) self.sink = sink or _log_sink self._proc: asyncio.subprocess.Process | None = None self._lock = asyncio.Lock() @@ -202,9 +215,19 @@ class MeloTTS: for s in segments ], "out": out, + "word_gap": self.word_gap, + "sentence_gap": self.sentence_gap, + "pitch": self.pitch, } else: # empty/whitespace reply: keep legacy single-utterance behaviour - payload = {"text": text, "out": out, "speed": self.speed} + payload = { + "text": text, + "out": out, + "speed": self.speed, + "word_gap": self.word_gap, + "sentence_gap": self.sentence_gap, + "pitch": self.pitch, + } req = json.dumps(payload) s = time.monotonic() async with self._lock: diff --git a/wsai/backends/melo_worker.py b/wsai/backends/melo_worker.py index ecc3ab9..158b69c 100644 --- a/wsai/backends/melo_worker.py +++ b/wsai/backends/melo_worker.py @@ -10,22 +10,34 @@ the original stdout carries the protocol, and fd 1 is redirected to fd 2 so all library chatter lands on stderr instead. Protocol (one JSON object per line, on the protocol channel): - <- {"text": "...", "out": "/abs/path.wav", "speed": 1.3} + <- {"text": "...", "out": "/abs/path.wav", "speed": 1.3, + "word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0} <- {"segments": [{"text": "...", "speed": 1.3, "pitch": 2.0}, ...], - "out": "/abs/path.wav"} # expressive form: per-segment speed + pitch + "out": "/abs/path.wav", + "word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0} -> {"ok": true, "out": "/abs/path.wav", "ms": 123} -> {"ok": false, "error": "..."} On startup, once the model is ready, it emits exactly one line: -> {"ready": true, "ms": , "device": "cpu"} -``pitch`` is a semitone offset applied to that segment's wav (0 == no shift) so -emotion tags can raise/lower the voice without changing the words. Segments are -synthesised independently and concatenated with a short gap so a single reply can -carry several emotions. +Four independent voice controls (ported from tts_site, see docs manual): + speed per-segment glyph speed -> generation-stage length_scale (1/speed) + word_gap reply-global, sec (-0.2..0.5): grow/shrink intra-sentence pauses + sentence_gap reply-global, sec (-0.5..1.5): insert/trim silence at sentence + boundaries (also used between emotion segments) + pitch reply-global semitone offset, added on top of each segment's own + ``pitch`` (from its emotion tag; 0 == no shift) + +Each segment's ``pitch`` is a semitone offset applied to that segment's wav so +emotion tags can raise/lower the voice without changing the words. A reply is +split into sentences, each sentence synthesised at its segment's speed, and the +pieces concatenated with ``sentence_gap`` so one reply can carry several emotions +and a controllable rhythm. """ import json import os +import re import sys import time @@ -79,7 +91,88 @@ def main() -> None: import numpy as np import soundfile - _GAP = np.zeros(int(sr * 0.12), dtype=np.float32) # 120 ms between segments + # ── 4-control synthesis, ported from tts_site (docs manual) ────────── + # glyph speed : generation-stage length_scale via tts_to_file(speed=) + # word_gap : scale the intra-sentence silences (sec, -0.2..0.5) + # sentence_gap : insert/trim silence at sentence boundaries (sec, -0.5..1.5) + # pitch : semitone shift (librosa), per-segment + global offset + _SENT_SPLIT_RE = re.compile(r"(?<=[.!?。!?…])\s+|\n+") + + def _clamp(v, lo, hi): + return float(max(lo, min(hi, float(v)))) + + def split_sentences(text: str) -> list[str]: + parts = [p.strip() for p in _SENT_SPLIT_RE.split(text) if p and p.strip()] + return parts or ([text.strip()] if text.strip() else []) + + def trim_end_silence(a, max_sec, thresh=0.02): + n = a.size + if n == 0 or max_sec <= 0: + return a, 0.0 + max_n = min(n, int(sr * max_sec)) + if max_n <= 0: + return a, 0.0 + tail = np.abs(a[n - max_n:]) + nz = np.where(tail >= thresh)[0] + cut = max_n if nz.size == 0 else (max_n - 1 - int(nz[-1])) + return (a, 0.0) if cut <= 0 else (a[: n - cut], cut / sr) + + def trim_start_silence(a, max_sec, thresh=0.02): + n = a.size + if n == 0 or max_sec <= 0: + return a, 0.0 + max_n = min(n, int(sr * max_sec)) + if max_n <= 0: + return a, 0.0 + head = np.abs(a[:max_n]) + nz = np.where(head >= thresh)[0] + cut = max_n if nz.size == 0 else int(nz[0]) + return (a, 0.0) if cut <= 0 else (a[cut:], cut / sr) + + def append_unit(pieces, unit, gap): + # gap >= 0: insert silence; gap < 0: trim boundary silence to tighten. + if not pieces: + pieces.append(unit) + return + if gap >= 0: + if gap > 1e-4: + pieces.append(np.zeros(int(sr * gap), dtype=np.float32)) + pieces.append(unit) + return + budget = -gap + prev, removed = trim_end_silence(pieces[-1], budget) + pieces[-1] = prev + budget -= removed + if budget > 1e-4: + unit, _ = trim_start_silence(unit, budget) + pieces.append(unit) + + def scale_word_gaps(a, delta, thresh=0.02, min_pause=0.08): + # Grow/shrink only the internal (non-boundary) silences of one sentence. + n = a.size + if abs(delta) < 1e-4 or n == 0: + return a + silent = np.abs(a) < thresh + changes = np.flatnonzero(np.diff(silent.astype(np.int8)) != 0) + 1 + bounds = [0, *changes.tolist(), n] + min_n = int(sr * min_pause) + floor_n = int(sr * 0.015) + add_n = int(delta * sr) + out, last = [], len(bounds) - 2 + for k in range(len(bounds) - 1): + s0, e0 = bounds[k], bounds[k + 1] + seg = a[s0:e0] + is_internal = 0 < k < last # keep leading/trailing boundary silence + if silent[s0] and is_internal and (e0 - s0) >= min_n: + new_n = max(floor_n, (e0 - s0) + add_n) + if new_n >= (e0 - s0): + seg = np.concatenate( + [seg, np.zeros(new_n - (e0 - s0), dtype=np.float32)] + ) + else: + seg = seg[:new_n] + out.append(seg) + return np.concatenate(out) if out else a def _pitch_shift(audio, semitones: float): if not semitones: @@ -90,20 +183,41 @@ def main() -> None: audio.astype(np.float32), sr=sr, n_steps=float(semitones) ) - def _synth_segments(segments: list[dict], out: str) -> None: - """Synthesize each segment, pitch-shift it, and concatenate to one wav.""" + def _synth_one(text, speed): + return np.asarray( + tts.tts_to_file(text, speaker_id, None, speed=speed), dtype=np.float32 + ) + + def _render(segments, out, word_gap, sentence_gap, pitch_offset): + """Render one wav from emotion segments applying the 4 controls. + + Each segment carries its own glyph ``speed`` and ``pitch`` (from emotion + tags); ``word_gap``/``sentence_gap``/``pitch_offset`` are reply-global. + Sentences within a segment, and the segments themselves, are joined with + ``sentence_gap`` (positive inserts silence, negative trims it).""" + word_gap = _clamp(word_gap, -0.2, 0.5) + sentence_gap = _clamp(sentence_gap, -0.5, 1.5) + pitch_offset = _clamp(pitch_offset, -12.0, 12.0) pieces = [] - for i, seg in enumerate(segments): - text = seg["text"] + for seg in segments: + text = seg.get("text", "") if not text.strip(): continue - speed = float(seg.get("speed", 1.0)) - pitch = float(seg.get("pitch", 0.0)) - audio = tts.tts_to_file(text, speaker_id, None, speed=speed) - audio = _pitch_shift(np.asarray(audio, dtype=np.float32), pitch) - if pieces: - pieces.append(_GAP) - pieces.append(audio) + speed = _clamp(seg.get("speed", 1.0), 0.5, 2.0) + pitch = _clamp(float(seg.get("pitch", 0.0)) + pitch_offset, -12.0, 12.0) + sent_pieces = [] + for sent in split_sentences(text): + a = _synth_one(sent, speed) + if abs(word_gap) > 1e-4: + a = scale_word_gaps(a, word_gap) + if not sent_pieces: + sent_pieces.append(a) + else: + append_unit(sent_pieces, a, sentence_gap) + if not sent_pieces: + continue + seg_audio = _pitch_shift(np.concatenate(sent_pieces), pitch) + append_unit(pieces, seg_audio, sentence_gap) if not pieces: raise ValueError("no speakable segment") soundfile.write(out, np.concatenate(pieces), sr) @@ -140,11 +254,14 @@ def main() -> None: if out.startswith("/tmp") or out.startswith("/dev/shm"): raise ValueError(f"refusing RAM-backed tmpfs path: {out}") s = time.monotonic() + word_gap = float(req.get("word_gap", 0.0)) + sentence_gap = float(req.get("sentence_gap", 0.0)) + pitch_offset = float(req.get("pitch", 0.0)) if "segments" in req: - _synth_segments(req["segments"], out) + _render(req["segments"], out, word_gap, sentence_gap, pitch_offset) else: # legacy single-utterance form - speed = float(req.get("speed", 1.0)) - tts.tts_to_file(req["text"], speaker_id, out, speed=speed) + seg = {"text": req["text"], "speed": float(req.get("speed", 1.0)), "pitch": 0.0} + _render([seg], out, word_gap, sentence_gap, pitch_offset) ms = int((time.monotonic() - s) * 1000) _emit({"ok": True, "out": out, "ms": ms}) except Exception as exc: # keep the worker alive across bad requests