"""Persistent MeloTTS worker (Korean). MeloTTS lives in its own Python (melo311); loading the model takes seconds, so we load it ONCE here and then serve synthesis requests over stdin/stdout. This process is launched with the melo311 interpreter by wsai.backends.melo.MeloTTS. MeloTTS (and its deps) print progress straight to stdout, which would corrupt the JSON protocol. So on startup we split the streams: a private duplicate of the original stdout carries the protocol, and fd 1 is redirected to fd 2 so all library chatter lands on stderr instead. Protocol (one JSON object per line, on the protocol channel): <- {"text": "...", "out": "/abs/path.wav", "speed": 1.3, "word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0} <- {"segments": [{"text": "...", "speed": 1.3, "pitch": 2.0}, ...], "out": "/abs/path.wav", "word_gap": 0.25, "sentence_gap": 0.75, "pitch": 0.0} -> {"ok": true, "out": "/abs/path.wav", "ms": 123} -> {"ok": false, "error": "..."} On startup, once the model is ready, it emits exactly one line: -> {"ready": true, "ms": , "device": "cpu"} Four independent voice controls (ported from tts_site, see docs manual): speed per-segment glyph speed -> generation-stage length_scale (1/speed) word_gap reply-global, sec (-0.2..0.5): grow/shrink intra-sentence pauses sentence_gap reply-global, sec (-0.5..1.5): insert/trim silence at sentence boundaries (also used between emotion segments) pitch reply-global semitone offset, added on top of each segment's own ``pitch`` (from its emotion tag; 0 == no shift) Each segment's ``pitch`` is a semitone offset applied to that segment's wav so emotion tags can raise/lower the voice without changing the words. A reply is split into sentences, each sentence synthesised at its segment's speed, and the pieces concatenated with ``sentence_gap`` so one reply can carry several emotions and a controllable rhythm. """ import json import os import re import sys import time # Split protocol from library noise BEFORE importing anything heavy. _proto = os.fdopen(os.dup(1), "w", buffering=1) # private copy of real stdout os.dup2(2, 1) # fd1 -> stderr, so stray library prints don't hit the protocol def _emit(obj: dict) -> None: _proto.write(json.dumps(obj) + "\n") _proto.flush() def _log(*a): print(*a, file=sys.stderr, flush=True) def main() -> None: lang = "KR" requested = os.environ.get("WSAI_MELO_DEVICE", "auto") # cpu | cuda | auto from melo.api import TTS # heavy import; only in the melo venv def _has_cuda() -> bool: try: import torch return torch.cuda.is_available() except Exception: return False device = requested if requested == "auto": device = "cuda" if _has_cuda() else "cpu" t0 = time.monotonic() try: tts = TTS(language=lang, device=device) except Exception as exc: # CUDA picked but unusable (CPU-only torch, missing libs, OOM): fall back # to CPU rather than leaving the whole voice loop dead. if device == "cuda": _log(f"[melo_worker] CUDA load failed ({exc}); falling back to CPU") device = "cpu" tts = TTS(language=lang, device=device) else: raise speaker_id = tts.hps.data.spk2id[lang] sr = tts.hps.data.sampling_rate load_ms = int((time.monotonic() - t0) * 1000) import numpy as np import soundfile # ── 4-control synthesis, ported from tts_site (docs manual) ────────── # glyph speed : generation-stage length_scale via tts_to_file(speed=) # word_gap : scale the intra-sentence silences (sec, -0.2..0.5) # sentence_gap : insert/trim silence at sentence boundaries (sec, -0.5..1.5) # pitch : semitone shift (librosa), per-segment + global offset _SENT_SPLIT_RE = re.compile(r"(?<=[.!?。!?…])\s+|\n+") def _clamp(v, lo, hi): return float(max(lo, min(hi, float(v)))) def split_sentences(text: str) -> list[str]: parts = [p.strip() for p in _SENT_SPLIT_RE.split(text) if p and p.strip()] return parts or ([text.strip()] if text.strip() else []) def trim_end_silence(a, max_sec, thresh=0.02): n = a.size if n == 0 or max_sec <= 0: return a, 0.0 max_n = min(n, int(sr * max_sec)) if max_n <= 0: return a, 0.0 tail = np.abs(a[n - max_n:]) nz = np.where(tail >= thresh)[0] cut = max_n if nz.size == 0 else (max_n - 1 - int(nz[-1])) return (a, 0.0) if cut <= 0 else (a[: n - cut], cut / sr) def trim_start_silence(a, max_sec, thresh=0.02): n = a.size if n == 0 or max_sec <= 0: return a, 0.0 max_n = min(n, int(sr * max_sec)) if max_n <= 0: return a, 0.0 head = np.abs(a[:max_n]) nz = np.where(head >= thresh)[0] cut = max_n if nz.size == 0 else int(nz[0]) return (a, 0.0) if cut <= 0 else (a[cut:], cut / sr) def append_unit(pieces, unit, gap): # gap >= 0: insert silence; gap < 0: trim boundary silence to tighten. if not pieces: pieces.append(unit) return if gap >= 0: if gap > 1e-4: pieces.append(np.zeros(int(sr * gap), dtype=np.float32)) pieces.append(unit) return budget = -gap prev, removed = trim_end_silence(pieces[-1], budget) pieces[-1] = prev budget -= removed if budget > 1e-4: unit, _ = trim_start_silence(unit, budget) pieces.append(unit) def scale_word_gaps(a, delta, thresh=0.02, min_pause=0.08): # Grow/shrink only the internal (non-boundary) silences of one sentence. n = a.size if abs(delta) < 1e-4 or n == 0: return a silent = np.abs(a) < thresh changes = np.flatnonzero(np.diff(silent.astype(np.int8)) != 0) + 1 bounds = [0, *changes.tolist(), n] min_n = int(sr * min_pause) floor_n = int(sr * 0.015) add_n = int(delta * sr) out, last = [], len(bounds) - 2 for k in range(len(bounds) - 1): s0, e0 = bounds[k], bounds[k + 1] seg = a[s0:e0] is_internal = 0 < k < last # keep leading/trailing boundary silence if silent[s0] and is_internal and (e0 - s0) >= min_n: new_n = max(floor_n, (e0 - s0) + add_n) if new_n >= (e0 - s0): seg = np.concatenate( [seg, np.zeros(new_n - (e0 - s0), dtype=np.float32)] ) else: seg = seg[:new_n] out.append(seg) return np.concatenate(out) if out else a def _pitch_shift(audio, semitones: float): if not semitones: return audio import librosa return librosa.effects.pitch_shift( audio.astype(np.float32), sr=sr, n_steps=float(semitones) ) def _synth_one(text, speed): return np.asarray( tts.tts_to_file(text, speaker_id, None, speed=speed), dtype=np.float32 ) def _render(segments, out, word_gap, sentence_gap, pitch_offset): """Render one wav from emotion segments applying the 4 controls. Each segment carries its own glyph ``speed`` and ``pitch`` (from emotion tags); ``word_gap``/``sentence_gap``/``pitch_offset`` are reply-global. Sentences within a segment, and the segments themselves, are joined with ``sentence_gap`` (positive inserts silence, negative trims it).""" word_gap = _clamp(word_gap, -0.2, 0.5) sentence_gap = _clamp(sentence_gap, -0.5, 1.5) pitch_offset = _clamp(pitch_offset, -12.0, 12.0) pieces = [] for seg in segments: text = seg.get("text", "") if not text.strip(): continue speed = _clamp(seg.get("speed", 1.0), 0.5, 2.0) pitch = _clamp(float(seg.get("pitch", 0.0)) + pitch_offset, -12.0, 12.0) sent_pieces = [] for sent in split_sentences(text): a = _synth_one(sent, speed) if abs(word_gap) > 1e-4: a = scale_word_gaps(a, word_gap) if not sent_pieces: sent_pieces.append(a) else: append_unit(sent_pieces, a, sentence_gap) if not sent_pieces: continue seg_audio = _pitch_shift(np.concatenate(sent_pieces), pitch) append_unit(pieces, seg_audio, sentence_gap) if not pieces: raise ValueError("no speakable segment") soundfile.write(out, np.concatenate(pieces), sr) # Warm up before signalling ready: the first CUDA synth pays a large lazy # cost (kernel autotune/cudnn), ~10s cold vs ~130ms hot, which would blow the # voice loop's ~1s budget on the very first reply. Do that dummy synth here so # "ready" means "hot". Failures must not block startup. warmup_ms = None try: warm_out = os.path.expanduser("~/.cache/wsai/tts/_warmup.wav") os.makedirs(os.path.dirname(warm_out), exist_ok=True) w = time.monotonic() tts.tts_to_file("워밍업", speaker_id, warm_out, speed=1.3) # Also JIT-warm librosa's pitch shifter (first call pays ~0.4s numba # compile) so the first *emotional* reply doesn't stall. import librosa librosa.effects.pitch_shift(np.zeros(sr, dtype=np.float32), sr=sr, n_steps=1.0) warmup_ms = int((time.monotonic() - w) * 1000) except Exception as exc: _log(f"[melo_worker] warmup skipped: {exc}") _emit({"ready": True, "ms": load_ms, "device": device, "warmup_ms": warmup_ms}) _log(f"[melo_worker] model ready in {load_ms} ms on {device} (warmup {warmup_ms} ms)") for line in sys.stdin: line = line.strip() if not line: continue try: req = json.loads(line) out = req["out"] if out.startswith("/tmp") or out.startswith("/dev/shm"): raise ValueError(f"refusing RAM-backed tmpfs path: {out}") s = time.monotonic() word_gap = float(req.get("word_gap", 0.0)) sentence_gap = float(req.get("sentence_gap", 0.0)) pitch_offset = float(req.get("pitch", 0.0)) if "segments" in req: _render(req["segments"], out, word_gap, sentence_gap, pitch_offset) else: # legacy single-utterance form seg = {"text": req["text"], "speed": float(req.get("speed", 1.0)), "pitch": 0.0} _render([seg], out, word_gap, sentence_gap, pitch_offset) ms = int((time.monotonic() - s) * 1000) _emit({"ok": True, "out": out, "ms": ms}) except Exception as exc: # keep the worker alive across bad requests _emit({"ok": False, "error": f"{type(exc).__name__}: {exc}"}) _log(f"[melo_worker] error: {exc}") if __name__ == "__main__": main()