Files
watch_sceen_ai/wsai/backends/melo_worker.py
EJClaw 0b92284ff8 feat(tts): port glyph-speed/word-gap/sentence-gap/pitch controls from tts_site
Ports the 4 independent voice controls documented in tts_site's manual into the
MeloTTS backend, layered on top of the existing emotion-tag segments:

- glyph speed: generation-stage length_scale (existing speed path); default
  bumped 1.2 -> 1.25 to match the manual (still well below the 1.5 that slurred)
- word_gap (sec, -0.2..0.5): scale intra-sentence silences after natural synth
- sentence_gap (sec, -0.5..1.5): insert/trim silence at sentence boundaries,
  replacing the old fixed 120ms inter-segment gap
- pitch (semitones, -12..12): global offset added on top of per-emotion pitch

Each reply is split into sentences, synthesised per-sentence at its segment's
speed, word_gap applied, joined with sentence_gap, then pitch-shifted. Exposed
via WSAI_TTS_SPEED/WORD_GAP/SENTENCE_GAP/PITCH env vars and MeloTTS ctor args.

Verified by real synthesis: each control independently changes wav duration
(speed, sentence_gap, word_gap-with-pauses, pitch); 29 tests pass.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-08-26 22:06:21 +09:00

274 lines
11 KiB
Python

"""Persistent MeloTTS worker (Korean).
MeloTTS lives in its own Python (melo311); loading the model takes seconds, so
we load it ONCE here and then serve synthesis requests over stdin/stdout. This
process is launched with the melo311 interpreter by wsai.backends.melo.MeloTTS.
MeloTTS (and its deps) print progress straight to stdout, which would corrupt
the JSON protocol. So on startup we split the streams: a private duplicate of
the original stdout carries the protocol, and fd 1 is redirected to fd 2 so all
library chatter lands on stderr instead.
Protocol (one JSON object per line, on the protocol channel):
<- {"text": "...", "out": "/abs/path.wav", "speed": 1.3,
"word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0}
<- {"segments": [{"text": "...", "speed": 1.3, "pitch": 2.0}, ...],
"out": "/abs/path.wav",
"word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0}
-> {"ok": true, "out": "/abs/path.wav", "ms": 123}
-> {"ok": false, "error": "..."}
On startup, once the model is ready, it emits exactly one line:
-> {"ready": true, "ms": <load-ms>, "device": "cpu"}
Four independent voice controls (ported from tts_site, see docs manual):
speed per-segment glyph speed -> generation-stage length_scale (1/speed)
word_gap reply-global, sec (-0.2..0.5): grow/shrink intra-sentence pauses
sentence_gap reply-global, sec (-0.5..1.5): insert/trim silence at sentence
boundaries (also used between emotion segments)
pitch reply-global semitone offset, added on top of each segment's own
``pitch`` (from its emotion tag; 0 == no shift)
Each segment's ``pitch`` is a semitone offset applied to that segment's wav so
emotion tags can raise/lower the voice without changing the words. A reply is
split into sentences, each sentence synthesised at its segment's speed, and the
pieces concatenated with ``sentence_gap`` so one reply can carry several emotions
and a controllable rhythm.
"""
import json
import os
import re
import sys
import time
# Split protocol from library noise BEFORE importing anything heavy.
_proto = os.fdopen(os.dup(1), "w", buffering=1) # private copy of real stdout
os.dup2(2, 1) # fd1 -> stderr, so stray library prints don't hit the protocol
def _emit(obj: dict) -> None:
_proto.write(json.dumps(obj) + "\n")
_proto.flush()
def _log(*a):
print(*a, file=sys.stderr, flush=True)
def main() -> None:
lang = "KR"
requested = os.environ.get("WSAI_MELO_DEVICE", "auto") # cpu | cuda | auto
from melo.api import TTS # heavy import; only in the melo venv
def _has_cuda() -> bool:
try:
import torch
return torch.cuda.is_available()
except Exception:
return False
device = requested
if requested == "auto":
device = "cuda" if _has_cuda() else "cpu"
t0 = time.monotonic()
try:
tts = TTS(language=lang, device=device)
except Exception as exc:
# CUDA picked but unusable (CPU-only torch, missing libs, OOM): fall back
# to CPU rather than leaving the whole voice loop dead.
if device == "cuda":
_log(f"[melo_worker] CUDA load failed ({exc}); falling back to CPU")
device = "cpu"
tts = TTS(language=lang, device=device)
else:
raise
speaker_id = tts.hps.data.spk2id[lang]
sr = tts.hps.data.sampling_rate
load_ms = int((time.monotonic() - t0) * 1000)
import numpy as np
import soundfile
# ── 4-control synthesis, ported from tts_site (docs manual) ──────────
# glyph speed : generation-stage length_scale via tts_to_file(speed=)
# word_gap : scale the intra-sentence silences (sec, -0.2..0.5)
# sentence_gap : insert/trim silence at sentence boundaries (sec, -0.5..1.5)
# pitch : semitone shift (librosa), per-segment + global offset
_SENT_SPLIT_RE = re.compile(r"(?<=[.!?。!?…])\s+|\n+")
def _clamp(v, lo, hi):
return float(max(lo, min(hi, float(v))))
def split_sentences(text: str) -> list[str]:
parts = [p.strip() for p in _SENT_SPLIT_RE.split(text) if p and p.strip()]
return parts or ([text.strip()] if text.strip() else [])
def trim_end_silence(a, max_sec, thresh=0.02):
n = a.size
if n == 0 or max_sec <= 0:
return a, 0.0
max_n = min(n, int(sr * max_sec))
if max_n <= 0:
return a, 0.0
tail = np.abs(a[n - max_n:])
nz = np.where(tail >= thresh)[0]
cut = max_n if nz.size == 0 else (max_n - 1 - int(nz[-1]))
return (a, 0.0) if cut <= 0 else (a[: n - cut], cut / sr)
def trim_start_silence(a, max_sec, thresh=0.02):
n = a.size
if n == 0 or max_sec <= 0:
return a, 0.0
max_n = min(n, int(sr * max_sec))
if max_n <= 0:
return a, 0.0
head = np.abs(a[:max_n])
nz = np.where(head >= thresh)[0]
cut = max_n if nz.size == 0 else int(nz[0])
return (a, 0.0) if cut <= 0 else (a[cut:], cut / sr)
def append_unit(pieces, unit, gap):
# gap >= 0: insert silence; gap < 0: trim boundary silence to tighten.
if not pieces:
pieces.append(unit)
return
if gap >= 0:
if gap > 1e-4:
pieces.append(np.zeros(int(sr * gap), dtype=np.float32))
pieces.append(unit)
return
budget = -gap
prev, removed = trim_end_silence(pieces[-1], budget)
pieces[-1] = prev
budget -= removed
if budget > 1e-4:
unit, _ = trim_start_silence(unit, budget)
pieces.append(unit)
def scale_word_gaps(a, delta, thresh=0.02, min_pause=0.08):
# Grow/shrink only the internal (non-boundary) silences of one sentence.
n = a.size
if abs(delta) < 1e-4 or n == 0:
return a
silent = np.abs(a) < thresh
changes = np.flatnonzero(np.diff(silent.astype(np.int8)) != 0) + 1
bounds = [0, *changes.tolist(), n]
min_n = int(sr * min_pause)
floor_n = int(sr * 0.015)
add_n = int(delta * sr)
out, last = [], len(bounds) - 2
for k in range(len(bounds) - 1):
s0, e0 = bounds[k], bounds[k + 1]
seg = a[s0:e0]
is_internal = 0 < k < last # keep leading/trailing boundary silence
if silent[s0] and is_internal and (e0 - s0) >= min_n:
new_n = max(floor_n, (e0 - s0) + add_n)
if new_n >= (e0 - s0):
seg = np.concatenate(
[seg, np.zeros(new_n - (e0 - s0), dtype=np.float32)]
)
else:
seg = seg[:new_n]
out.append(seg)
return np.concatenate(out) if out else a
def _pitch_shift(audio, semitones: float):
if not semitones:
return audio
import librosa
return librosa.effects.pitch_shift(
audio.astype(np.float32), sr=sr, n_steps=float(semitones)
)
def _synth_one(text, speed):
return np.asarray(
tts.tts_to_file(text, speaker_id, None, speed=speed), dtype=np.float32
)
def _render(segments, out, word_gap, sentence_gap, pitch_offset):
"""Render one wav from emotion segments applying the 4 controls.
Each segment carries its own glyph ``speed`` and ``pitch`` (from emotion
tags); ``word_gap``/``sentence_gap``/``pitch_offset`` are reply-global.
Sentences within a segment, and the segments themselves, are joined with
``sentence_gap`` (positive inserts silence, negative trims it)."""
word_gap = _clamp(word_gap, -0.2, 0.5)
sentence_gap = _clamp(sentence_gap, -0.5, 1.5)
pitch_offset = _clamp(pitch_offset, -12.0, 12.0)
pieces = []
for seg in segments:
text = seg.get("text", "")
if not text.strip():
continue
speed = _clamp(seg.get("speed", 1.0), 0.5, 2.0)
pitch = _clamp(float(seg.get("pitch", 0.0)) + pitch_offset, -12.0, 12.0)
sent_pieces = []
for sent in split_sentences(text):
a = _synth_one(sent, speed)
if abs(word_gap) > 1e-4:
a = scale_word_gaps(a, word_gap)
if not sent_pieces:
sent_pieces.append(a)
else:
append_unit(sent_pieces, a, sentence_gap)
if not sent_pieces:
continue
seg_audio = _pitch_shift(np.concatenate(sent_pieces), pitch)
append_unit(pieces, seg_audio, sentence_gap)
if not pieces:
raise ValueError("no speakable segment")
soundfile.write(out, np.concatenate(pieces), sr)
# Warm up before signalling ready: the first CUDA synth pays a large lazy
# cost (kernel autotune/cudnn), ~10s cold vs ~130ms hot, which would blow the
# voice loop's ~1s budget on the very first reply. Do that dummy synth here so
# "ready" means "hot". Failures must not block startup.
warmup_ms = None
try:
warm_out = os.path.expanduser("~/.cache/wsai/tts/_warmup.wav")
os.makedirs(os.path.dirname(warm_out), exist_ok=True)
w = time.monotonic()
tts.tts_to_file("워밍업", speaker_id, warm_out, speed=1.3)
# Also JIT-warm librosa's pitch shifter (first call pays ~0.4s numba
# compile) so the first *emotional* reply doesn't stall.
import librosa
librosa.effects.pitch_shift(np.zeros(sr, dtype=np.float32), sr=sr, n_steps=1.0)
warmup_ms = int((time.monotonic() - w) * 1000)
except Exception as exc:
_log(f"[melo_worker] warmup skipped: {exc}")
_emit({"ready": True, "ms": load_ms, "device": device, "warmup_ms": warmup_ms})
_log(f"[melo_worker] model ready in {load_ms} ms on {device} (warmup {warmup_ms} ms)")
for line in sys.stdin:
line = line.strip()
if not line:
continue
try:
req = json.loads(line)
out = req["out"]
if out.startswith("/tmp") or out.startswith("/dev/shm"):
raise ValueError(f"refusing RAM-backed tmpfs path: {out}")
s = time.monotonic()
word_gap = float(req.get("word_gap", 0.0))
sentence_gap = float(req.get("sentence_gap", 0.0))
pitch_offset = float(req.get("pitch", 0.0))
if "segments" in req:
_render(req["segments"], out, word_gap, sentence_gap, pitch_offset)
else: # legacy single-utterance form
seg = {"text": req["text"], "speed": float(req.get("speed", 1.0)), "pitch": 0.0}
_render([seg], out, word_gap, sentence_gap, pitch_offset)
ms = int((time.monotonic() - s) * 1000)
_emit({"ok": True, "out": out, "ms": ms})
except Exception as exc: # keep the worker alive across bad requests
_emit({"ok": False, "error": f"{type(exc).__name__}: {exc}"})
_log(f"[melo_worker] error: {exc}")
if __name__ == "__main__":
main()