- 타임스트레치 우선순위를 ffmpeg atempo(기본) → rubberband → librosa 로 변경 - MeloEngine: 문장 단위로 분리 후 문장 사이 무음(sentence_gap), 단어(어절) 사이 무음(word_gap)을 타임스트레치와 독립적으로 삽입. speed 는 글자(음소) 속도만 조절 - API(TTSRequest)에 word_gap/sentence_gap 필드 추가, 전 엔진 시그니처 확장 - 프론트엔드에 '글자 속도/단어 간격/문장 간격' 슬라이더 추가, supports 플래그로 미지원 엔진(supertonic/oute/cosy)에서는 간격 슬라이더 자동 비활성화
55 lines
1.6 KiB
Python
55 lines
1.6 KiB
Python
"""TTS 엔진 공통 인터페이스.
|
|
|
|
각 엔진은 서로 다른 의존성(torch/librosa/numpy 핀)을 가질 수 있으므로,
|
|
in-process 로 도는 엔진(MeloTTS)과 전용 venv 를 subprocess 로 호출하는 엔진
|
|
(Coqui 등)을 동일한 인터페이스로 노출한다.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from abc import ABC, abstractmethod
|
|
|
|
|
|
class BaseEngine(ABC):
|
|
id: str = ""
|
|
label: str = ""
|
|
license: str = ""
|
|
uses_gpu: bool = False
|
|
notes: str = ""
|
|
|
|
@abstractmethod
|
|
def describe(self) -> dict:
|
|
"""엔진 메타데이터 + 언어/목소리 목록.
|
|
|
|
반환 예:
|
|
{
|
|
"id": "melo", "label": "MeloTTS", "license": "MIT",
|
|
"uses_gpu": True, "notes": "...",
|
|
"languages": [ {"code": "KR", "label": "한국어",
|
|
"speakers": [{"id": "KR", "label": "기본"}]} ],
|
|
"supports": {"speed": True, "pitch": True},
|
|
}
|
|
"""
|
|
raise NotImplementedError
|
|
|
|
@abstractmethod
|
|
def synth_wav(
|
|
self,
|
|
text: str,
|
|
language: str,
|
|
speaker: str | None = None,
|
|
speed: float = 1.0,
|
|
pitch: float = 0.0,
|
|
word_gap: float = 0.0,
|
|
sentence_gap: float | None = None,
|
|
) -> bytes:
|
|
"""WAV(PCM16) 바이트를 반환.
|
|
|
|
speed=글자(음소) 속도, word_gap=단어 사이 무음(초),
|
|
sentence_gap=문장 사이 무음(초). 간격을 지원하지 않는 엔진은 무시한다.
|
|
"""
|
|
raise NotImplementedError
|
|
|
|
def warmup(self) -> None:
|
|
"""선택적: 서버 시작 시 모델 프리로드."""
|
|
return None
|