- 타임스트레치 우선순위를 ffmpeg atempo(기본) → rubberband → librosa 로 변경 - MeloEngine: 문장 단위로 분리 후 문장 사이 무음(sentence_gap), 단어(어절) 사이 무음(word_gap)을 타임스트레치와 독립적으로 삽입. speed 는 글자(음소) 속도만 조절 - API(TTSRequest)에 word_gap/sentence_gap 필드 추가, 전 엔진 시그니처 확장 - 프론트엔드에 '글자 속도/단어 간격/문장 간격' 슬라이더 추가, supports 플래그로 미지원 엔진(supertonic/oute/cosy)에서는 간격 슬라이더 자동 비활성화
102 lines
3.2 KiB
Python
102 lines
3.2 KiB
Python
"""Supertonic 엔진 (프록시).
|
|
|
|
실제 합성은 전용 venv 워커(worker.py)가 담당하고, 이 엔진은 HTTP로 요청을
|
|
전달한다. 워커 URL(SUPERTONIC_URL)이 설정된 경우에만 활성화된다.
|
|
자연스러운 한국어 음성(수퍼톤), ONNX 기반이라 GPU 불필요.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
import urllib.error
|
|
import urllib.request
|
|
|
|
from .base import BaseEngine
|
|
|
|
WORKER_URL = os.environ.get("SUPERTONIC_URL", "").rstrip("/")
|
|
|
|
_VOICES = [
|
|
{"id": "F1", "label": "여성 1"},
|
|
{"id": "F2", "label": "여성 2"},
|
|
{"id": "F3", "label": "여성 3"},
|
|
{"id": "F4", "label": "여성 4"},
|
|
{"id": "F5", "label": "여성 5"},
|
|
{"id": "M1", "label": "남성 1"},
|
|
{"id": "M2", "label": "남성 2"},
|
|
{"id": "M3", "label": "남성 3"},
|
|
{"id": "M4", "label": "남성 4"},
|
|
{"id": "M5", "label": "남성 5"},
|
|
]
|
|
|
|
|
|
class SupertonicEngine(BaseEngine):
|
|
id = "supertonic"
|
|
label = "Supertonic (자연 음성)"
|
|
license = "MIT"
|
|
uses_gpu = False
|
|
notes = "수퍼톤의 자연스러운 다국어 음성. 한국어/자동감지, 남녀 10종, 속도·피치."
|
|
|
|
@staticmethod
|
|
def available() -> bool:
|
|
return bool(WORKER_URL)
|
|
|
|
def describe(self) -> dict:
|
|
return {
|
|
"id": self.id,
|
|
"label": self.label,
|
|
"license": self.license,
|
|
"uses_gpu": self.uses_gpu,
|
|
"notes": self.notes,
|
|
"languages": [
|
|
{"code": "KO", "label": "한국어", "speakers": _VOICES},
|
|
{"code": "AUTO", "label": "자동 감지 (다국어)", "speakers": _VOICES},
|
|
],
|
|
"default_language": "KO",
|
|
"supports": {
|
|
"speed": True,
|
|
"pitch": True,
|
|
"word_gap": False,
|
|
"sentence_gap": False,
|
|
},
|
|
}
|
|
|
|
def synth_wav(
|
|
self,
|
|
text: str,
|
|
language: str,
|
|
speaker: str | None = None,
|
|
speed: float = 1.0,
|
|
pitch: float = 0.0,
|
|
word_gap: float = 0.0,
|
|
sentence_gap: float | None = None,
|
|
) -> bytes:
|
|
text = (text or "").strip()
|
|
if not text:
|
|
raise ValueError("텍스트가 비어 있습니다.")
|
|
if not WORKER_URL:
|
|
raise ValueError("Supertonic 워커가 설정되지 않았습니다.")
|
|
|
|
lang = "na" if (language or "KO").upper() == "AUTO" else "ko"
|
|
payload = json.dumps(
|
|
{
|
|
"text": text,
|
|
"voice": speaker or "F1",
|
|
"lang": lang,
|
|
"speed": speed,
|
|
"pitch": pitch,
|
|
}
|
|
).encode("utf-8")
|
|
request = urllib.request.Request(
|
|
WORKER_URL + "/synth",
|
|
data=payload,
|
|
headers={"Content-Type": "application/json"},
|
|
)
|
|
try:
|
|
with urllib.request.urlopen(request, timeout=120) as resp:
|
|
return resp.read()
|
|
except urllib.error.HTTPError as exc:
|
|
detail = exc.read().decode("utf-8", errors="ignore")
|
|
raise ValueError(f"Supertonic 합성 실패: {detail}")
|
|
except Exception as exc: # pragma: no cover
|
|
raise ValueError(f"Supertonic 워커 연결 실패: {exc}")
|