Ports the 4 independent voice controls documented in tts_site's manual into the MeloTTS backend, layered on top of the existing emotion-tag segments: - glyph speed: generation-stage length_scale (existing speed path); default bumped 1.2 -> 1.25 to match the manual (still well below the 1.5 that slurred) - word_gap (sec, -0.2..0.5): scale intra-sentence silences after natural synth - sentence_gap (sec, -0.5..1.5): insert/trim silence at sentence boundaries, replacing the old fixed 120ms inter-segment gap - pitch (semitones, -12..12): global offset added on top of per-emotion pitch Each reply is split into sentences, synthesised per-sentence at its segment's speed, word_gap applied, joined with sentence_gap, then pitch-shifted. Exposed via WSAI_TTS_SPEED/WORD_GAP/SENTENCE_GAP/PITCH env vars and MeloTTS ctor args. Verified by real synthesis: each control independently changes wav duration (speed, sentence_gap, word_gap-with-pauses, pitch); 29 tests pass. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
274 lines
11 KiB
Python
274 lines
11 KiB
Python
"""Persistent MeloTTS worker (Korean).
|
|
|
|
MeloTTS lives in its own Python (melo311); loading the model takes seconds, so
|
|
we load it ONCE here and then serve synthesis requests over stdin/stdout. This
|
|
process is launched with the melo311 interpreter by wsai.backends.melo.MeloTTS.
|
|
|
|
MeloTTS (and its deps) print progress straight to stdout, which would corrupt
|
|
the JSON protocol. So on startup we split the streams: a private duplicate of
|
|
the original stdout carries the protocol, and fd 1 is redirected to fd 2 so all
|
|
library chatter lands on stderr instead.
|
|
|
|
Protocol (one JSON object per line, on the protocol channel):
|
|
<- {"text": "...", "out": "/abs/path.wav", "speed": 1.3,
|
|
"word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0}
|
|
<- {"segments": [{"text": "...", "speed": 1.3, "pitch": 2.0}, ...],
|
|
"out": "/abs/path.wav",
|
|
"word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0}
|
|
-> {"ok": true, "out": "/abs/path.wav", "ms": 123}
|
|
-> {"ok": false, "error": "..."}
|
|
On startup, once the model is ready, it emits exactly one line:
|
|
-> {"ready": true, "ms": <load-ms>, "device": "cpu"}
|
|
|
|
Four independent voice controls (ported from tts_site, see docs manual):
|
|
speed per-segment glyph speed -> generation-stage length_scale (1/speed)
|
|
word_gap reply-global, sec (-0.2..0.5): grow/shrink intra-sentence pauses
|
|
sentence_gap reply-global, sec (-0.5..1.5): insert/trim silence at sentence
|
|
boundaries (also used between emotion segments)
|
|
pitch reply-global semitone offset, added on top of each segment's own
|
|
``pitch`` (from its emotion tag; 0 == no shift)
|
|
|
|
Each segment's ``pitch`` is a semitone offset applied to that segment's wav so
|
|
emotion tags can raise/lower the voice without changing the words. A reply is
|
|
split into sentences, each sentence synthesised at its segment's speed, and the
|
|
pieces concatenated with ``sentence_gap`` so one reply can carry several emotions
|
|
and a controllable rhythm.
|
|
"""
|
|
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
import time
|
|
|
|
# Split protocol from library noise BEFORE importing anything heavy.
|
|
_proto = os.fdopen(os.dup(1), "w", buffering=1) # private copy of real stdout
|
|
os.dup2(2, 1) # fd1 -> stderr, so stray library prints don't hit the protocol
|
|
|
|
|
|
def _emit(obj: dict) -> None:
|
|
_proto.write(json.dumps(obj) + "\n")
|
|
_proto.flush()
|
|
|
|
|
|
def _log(*a):
|
|
print(*a, file=sys.stderr, flush=True)
|
|
|
|
|
|
def main() -> None:
|
|
lang = "KR"
|
|
requested = os.environ.get("WSAI_MELO_DEVICE", "auto") # cpu | cuda | auto
|
|
from melo.api import TTS # heavy import; only in the melo venv
|
|
|
|
def _has_cuda() -> bool:
|
|
try:
|
|
import torch
|
|
|
|
return torch.cuda.is_available()
|
|
except Exception:
|
|
return False
|
|
|
|
device = requested
|
|
if requested == "auto":
|
|
device = "cuda" if _has_cuda() else "cpu"
|
|
|
|
t0 = time.monotonic()
|
|
try:
|
|
tts = TTS(language=lang, device=device)
|
|
except Exception as exc:
|
|
# CUDA picked but unusable (CPU-only torch, missing libs, OOM): fall back
|
|
# to CPU rather than leaving the whole voice loop dead.
|
|
if device == "cuda":
|
|
_log(f"[melo_worker] CUDA load failed ({exc}); falling back to CPU")
|
|
device = "cpu"
|
|
tts = TTS(language=lang, device=device)
|
|
else:
|
|
raise
|
|
speaker_id = tts.hps.data.spk2id[lang]
|
|
sr = tts.hps.data.sampling_rate
|
|
load_ms = int((time.monotonic() - t0) * 1000)
|
|
|
|
import numpy as np
|
|
import soundfile
|
|
|
|
# ── 4-control synthesis, ported from tts_site (docs manual) ──────────
|
|
# glyph speed : generation-stage length_scale via tts_to_file(speed=)
|
|
# word_gap : scale the intra-sentence silences (sec, -0.2..0.5)
|
|
# sentence_gap : insert/trim silence at sentence boundaries (sec, -0.5..1.5)
|
|
# pitch : semitone shift (librosa), per-segment + global offset
|
|
_SENT_SPLIT_RE = re.compile(r"(?<=[.!?。!?…])\s+|\n+")
|
|
|
|
def _clamp(v, lo, hi):
|
|
return float(max(lo, min(hi, float(v))))
|
|
|
|
def split_sentences(text: str) -> list[str]:
|
|
parts = [p.strip() for p in _SENT_SPLIT_RE.split(text) if p and p.strip()]
|
|
return parts or ([text.strip()] if text.strip() else [])
|
|
|
|
def trim_end_silence(a, max_sec, thresh=0.02):
|
|
n = a.size
|
|
if n == 0 or max_sec <= 0:
|
|
return a, 0.0
|
|
max_n = min(n, int(sr * max_sec))
|
|
if max_n <= 0:
|
|
return a, 0.0
|
|
tail = np.abs(a[n - max_n:])
|
|
nz = np.where(tail >= thresh)[0]
|
|
cut = max_n if nz.size == 0 else (max_n - 1 - int(nz[-1]))
|
|
return (a, 0.0) if cut <= 0 else (a[: n - cut], cut / sr)
|
|
|
|
def trim_start_silence(a, max_sec, thresh=0.02):
|
|
n = a.size
|
|
if n == 0 or max_sec <= 0:
|
|
return a, 0.0
|
|
max_n = min(n, int(sr * max_sec))
|
|
if max_n <= 0:
|
|
return a, 0.0
|
|
head = np.abs(a[:max_n])
|
|
nz = np.where(head >= thresh)[0]
|
|
cut = max_n if nz.size == 0 else int(nz[0])
|
|
return (a, 0.0) if cut <= 0 else (a[cut:], cut / sr)
|
|
|
|
def append_unit(pieces, unit, gap):
|
|
# gap >= 0: insert silence; gap < 0: trim boundary silence to tighten.
|
|
if not pieces:
|
|
pieces.append(unit)
|
|
return
|
|
if gap >= 0:
|
|
if gap > 1e-4:
|
|
pieces.append(np.zeros(int(sr * gap), dtype=np.float32))
|
|
pieces.append(unit)
|
|
return
|
|
budget = -gap
|
|
prev, removed = trim_end_silence(pieces[-1], budget)
|
|
pieces[-1] = prev
|
|
budget -= removed
|
|
if budget > 1e-4:
|
|
unit, _ = trim_start_silence(unit, budget)
|
|
pieces.append(unit)
|
|
|
|
def scale_word_gaps(a, delta, thresh=0.02, min_pause=0.08):
|
|
# Grow/shrink only the internal (non-boundary) silences of one sentence.
|
|
n = a.size
|
|
if abs(delta) < 1e-4 or n == 0:
|
|
return a
|
|
silent = np.abs(a) < thresh
|
|
changes = np.flatnonzero(np.diff(silent.astype(np.int8)) != 0) + 1
|
|
bounds = [0, *changes.tolist(), n]
|
|
min_n = int(sr * min_pause)
|
|
floor_n = int(sr * 0.015)
|
|
add_n = int(delta * sr)
|
|
out, last = [], len(bounds) - 2
|
|
for k in range(len(bounds) - 1):
|
|
s0, e0 = bounds[k], bounds[k + 1]
|
|
seg = a[s0:e0]
|
|
is_internal = 0 < k < last # keep leading/trailing boundary silence
|
|
if silent[s0] and is_internal and (e0 - s0) >= min_n:
|
|
new_n = max(floor_n, (e0 - s0) + add_n)
|
|
if new_n >= (e0 - s0):
|
|
seg = np.concatenate(
|
|
[seg, np.zeros(new_n - (e0 - s0), dtype=np.float32)]
|
|
)
|
|
else:
|
|
seg = seg[:new_n]
|
|
out.append(seg)
|
|
return np.concatenate(out) if out else a
|
|
|
|
def _pitch_shift(audio, semitones: float):
|
|
if not semitones:
|
|
return audio
|
|
import librosa
|
|
|
|
return librosa.effects.pitch_shift(
|
|
audio.astype(np.float32), sr=sr, n_steps=float(semitones)
|
|
)
|
|
|
|
def _synth_one(text, speed):
|
|
return np.asarray(
|
|
tts.tts_to_file(text, speaker_id, None, speed=speed), dtype=np.float32
|
|
)
|
|
|
|
def _render(segments, out, word_gap, sentence_gap, pitch_offset):
|
|
"""Render one wav from emotion segments applying the 4 controls.
|
|
|
|
Each segment carries its own glyph ``speed`` and ``pitch`` (from emotion
|
|
tags); ``word_gap``/``sentence_gap``/``pitch_offset`` are reply-global.
|
|
Sentences within a segment, and the segments themselves, are joined with
|
|
``sentence_gap`` (positive inserts silence, negative trims it)."""
|
|
word_gap = _clamp(word_gap, -0.2, 0.5)
|
|
sentence_gap = _clamp(sentence_gap, -0.5, 1.5)
|
|
pitch_offset = _clamp(pitch_offset, -12.0, 12.0)
|
|
pieces = []
|
|
for seg in segments:
|
|
text = seg.get("text", "")
|
|
if not text.strip():
|
|
continue
|
|
speed = _clamp(seg.get("speed", 1.0), 0.5, 2.0)
|
|
pitch = _clamp(float(seg.get("pitch", 0.0)) + pitch_offset, -12.0, 12.0)
|
|
sent_pieces = []
|
|
for sent in split_sentences(text):
|
|
a = _synth_one(sent, speed)
|
|
if abs(word_gap) > 1e-4:
|
|
a = scale_word_gaps(a, word_gap)
|
|
if not sent_pieces:
|
|
sent_pieces.append(a)
|
|
else:
|
|
append_unit(sent_pieces, a, sentence_gap)
|
|
if not sent_pieces:
|
|
continue
|
|
seg_audio = _pitch_shift(np.concatenate(sent_pieces), pitch)
|
|
append_unit(pieces, seg_audio, sentence_gap)
|
|
if not pieces:
|
|
raise ValueError("no speakable segment")
|
|
soundfile.write(out, np.concatenate(pieces), sr)
|
|
|
|
# Warm up before signalling ready: the first CUDA synth pays a large lazy
|
|
# cost (kernel autotune/cudnn), ~10s cold vs ~130ms hot, which would blow the
|
|
# voice loop's ~1s budget on the very first reply. Do that dummy synth here so
|
|
# "ready" means "hot". Failures must not block startup.
|
|
warmup_ms = None
|
|
try:
|
|
warm_out = os.path.expanduser("~/.cache/wsai/tts/_warmup.wav")
|
|
os.makedirs(os.path.dirname(warm_out), exist_ok=True)
|
|
w = time.monotonic()
|
|
tts.tts_to_file("워밍업", speaker_id, warm_out, speed=1.3)
|
|
# Also JIT-warm librosa's pitch shifter (first call pays ~0.4s numba
|
|
# compile) so the first *emotional* reply doesn't stall.
|
|
import librosa
|
|
|
|
librosa.effects.pitch_shift(np.zeros(sr, dtype=np.float32), sr=sr, n_steps=1.0)
|
|
warmup_ms = int((time.monotonic() - w) * 1000)
|
|
except Exception as exc:
|
|
_log(f"[melo_worker] warmup skipped: {exc}")
|
|
|
|
_emit({"ready": True, "ms": load_ms, "device": device, "warmup_ms": warmup_ms})
|
|
_log(f"[melo_worker] model ready in {load_ms} ms on {device} (warmup {warmup_ms} ms)")
|
|
|
|
for line in sys.stdin:
|
|
line = line.strip()
|
|
if not line:
|
|
continue
|
|
try:
|
|
req = json.loads(line)
|
|
out = req["out"]
|
|
if out.startswith("/tmp") or out.startswith("/dev/shm"):
|
|
raise ValueError(f"refusing RAM-backed tmpfs path: {out}")
|
|
s = time.monotonic()
|
|
word_gap = float(req.get("word_gap", 0.0))
|
|
sentence_gap = float(req.get("sentence_gap", 0.0))
|
|
pitch_offset = float(req.get("pitch", 0.0))
|
|
if "segments" in req:
|
|
_render(req["segments"], out, word_gap, sentence_gap, pitch_offset)
|
|
else: # legacy single-utterance form
|
|
seg = {"text": req["text"], "speed": float(req.get("speed", 1.0)), "pitch": 0.0}
|
|
_render([seg], out, word_gap, sentence_gap, pitch_offset)
|
|
ms = int((time.monotonic() - s) * 1000)
|
|
_emit({"ok": True, "out": out, "ms": ms})
|
|
except Exception as exc: # keep the worker alive across bad requests
|
|
_emit({"ok": False, "error": f"{type(exc).__name__}: {exc}"})
|
|
_log(f"[melo_worker] error: {exc}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|