Files
tts_site/app/server.py
claude c0665893a7 feat: 글자속도·단어간격·문장간격 독립 조절 + atempo 기본 타임스트레치
- 타임스트레치 우선순위를 ffmpeg atempo(기본) → rubberband → librosa 로 변경
- MeloEngine: 문장 단위로 분리 후 문장 사이 무음(sentence_gap), 단어(어절) 사이
  무음(word_gap)을 타임스트레치와 독립적으로 삽입. speed 는 글자(음소) 속도만 조절
- API(TTSRequest)에 word_gap/sentence_gap 필드 추가, 전 엔진 시그니처 확장
- 프론트엔드에 '글자 속도/단어 간격/문장 간격' 슬라이더 추가, supports 플래그로
  미지원 엔진(supertonic/oute/cosy)에서는 간격 슬라이더 자동 비활성화
2026-08-25 23:54:48 +09:00

106 lines
3.0 KiB
Python

"""한국어 우선 다중 엔진 TTS 웹 서비스.
- 로그인/과금 없이 무제한 사용
- 엔진(MeloTTS, Coqui GlowTTS-KSS 등) 선택 → 언어/목소리 선택
- 속도 / 피치 조절
- GPU 자동 사용
"""
from __future__ import annotations
import os
from fastapi import FastAPI, HTTPException
from fastapi.middleware.cors import CORSMiddleware
from fastapi.responses import FileResponse, Response
from fastapi.staticfiles import StaticFiles
from pydantic import BaseModel, Field
from .engines import (
default_engine_id,
get_engine,
list_engines,
warmup_all,
)
FRONTEND_DIR = os.path.join(os.path.dirname(os.path.dirname(__file__)), "frontend")
app = FastAPI(title="한국어 TTS 스튜디오", version="2.0.0")
app.add_middleware(
CORSMiddleware,
allow_origins=["*"],
allow_methods=["*"],
allow_headers=["*"],
)
class TTSRequest(BaseModel):
text: str = Field(..., description="읽을 텍스트")
engine: str | None = Field(None, description="엔진 ID (melo / coqui-kss 등)")
language: str = Field("KR", description="언어 코드")
speaker: str | None = Field(None, description="화자/모델 ID")
speed: float = Field(1.4, ge=0.5, le=2.0, description="글자(음소) 발화 속도")
pitch: float = Field(0.0, ge=-12.0, le=12.0, description="피치(반음)")
word_gap: float = Field(0.0, ge=0.0, le=1.0, description="단어 사이 무음(초)")
sentence_gap: float = Field(0.3, ge=0.0, le=2.0, description="문장 사이 무음(초)")
@app.on_event("startup")
def _startup() -> None:
warmup_all()
@app.get("/api/health")
def health() -> dict:
import torch
return {
"status": "ok",
"cuda": torch.cuda.is_available(),
"gpu": torch.cuda.get_device_name(0) if torch.cuda.is_available() else None,
"engines": [e.id for e in list_engines()],
"default_engine": default_engine_id(),
}
@app.get("/api/engines")
def engines() -> dict:
return {
"default": default_engine_id(),
"engines": [e.describe() for e in list_engines()],
}
@app.post("/api/tts")
def tts(req: TTSRequest) -> Response:
engine_id = req.engine or default_engine_id()
try:
engine = get_engine(engine_id)
wav = engine.synth_wav(
text=req.text,
language=req.language,
speaker=req.speaker,
speed=req.speed,
pitch=req.pitch,
word_gap=req.word_gap,
sentence_gap=req.sentence_gap,
)
except ValueError as exc:
raise HTTPException(status_code=400, detail=str(exc))
except Exception as exc: # pragma: no cover
raise HTTPException(status_code=500, detail=f"합성 실패: {exc}")
return Response(
content=wav,
media_type="audio/wav",
headers={"Content-Disposition": 'inline; filename="tts.wav"'},
)
@app.get("/")
def index() -> FileResponse:
return FileResponse(os.path.join(FRONTEND_DIR, "index.html"))
app.mount("/static", StaticFiles(directory=FRONTEND_DIR), name="static")