feat: CosyVoice 한국어 엔진 통합 (Apache-2.0, GPU)
- vendor/CosyVoice: 패치된 CosyVoice 코드 + Matcha-TTS 벤더링 - cosyvoice_worker: 전용 venv에서 상주하는 합성 워커(FastAPI, 모델 1회 로드) - app/engines/cosyvoice_engine: HTTP 프록시 엔진(속도/피치) - Dockerfile: 전용 venv(torch cu128) + 모델 베이킹 + 빌드 워밍업 - entrypoint.sh: 워커(8001) + 메인앱(8788) 동시 기동
This commit is contained in:
35
cosyvoice_worker/requirements.txt
Normal file
35
cosyvoice_worker/requirements.txt
Normal file
@@ -0,0 +1,35 @@
|
||||
# CosyVoice 워커 전용 의존성 (호스트에서 검증된 조합).
|
||||
# torch/torchaudio 는 Dockerfile 에서 cu128 휠로 별도 설치.
|
||||
# openai-whisper/tiktoken/more-itertools 는 --no-deps 로 별도 설치(토치 덮어쓰기 방지).
|
||||
# pyworld/tensorrt/deepspeed/onnxruntime-gpu 는 추론에 불필요하여 제외.
|
||||
conformer==0.3.2
|
||||
diffusers==0.29.0
|
||||
fastapi==0.115.6
|
||||
fastapi-cli==0.0.4
|
||||
gdown==5.1.0
|
||||
gradio==5.4.0
|
||||
grpcio==1.57.0
|
||||
grpcio-tools==1.57.0
|
||||
hydra-core==1.3.2
|
||||
HyperPyYAML==1.2.3
|
||||
inflect==7.3.1
|
||||
librosa==0.10.2
|
||||
lightning==2.2.4
|
||||
matplotlib==3.7.5
|
||||
modelscope==1.20.0
|
||||
networkx==3.1
|
||||
numpy==1.26.4
|
||||
omegaconf==2.3.0
|
||||
onnx==1.16.0
|
||||
onnxruntime==1.18.0
|
||||
protobuf==4.25
|
||||
pyarrow==18.1.0
|
||||
pydantic==2.7.0
|
||||
rich==13.7.1
|
||||
soundfile==0.12.1
|
||||
tensorboard==2.14.0
|
||||
transformers==4.51.3
|
||||
x-transformers==2.11.24
|
||||
uvicorn==0.30.0
|
||||
wetext==0.0.4
|
||||
wget==3.2
|
||||
105
cosyvoice_worker/worker.py
Normal file
105
cosyvoice_worker/worker.py
Normal file
@@ -0,0 +1,105 @@
|
||||
"""CosyVoice 한국어 합성 워커 (전용 venv에서 실행).
|
||||
|
||||
모델을 1회만 로드해 상주시키고, 메인 앱이 HTTP로 합성을 요청한다.
|
||||
- SFT 프리셋 화자('韩语女' = 한국어)로 참조음성 없이 합성
|
||||
- 속도(speed)는 CosyVoice 네이티브, 피치(pitch)는 librosa 후처리
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import os
|
||||
import sys
|
||||
|
||||
# 벤더링된 CosyVoice 코드 경로
|
||||
sys.path.insert(0, "/opt/CosyVoice")
|
||||
sys.path.insert(0, "/opt/CosyVoice/third_party/Matcha-TTS")
|
||||
|
||||
import numpy as np
|
||||
import soundfile as sf
|
||||
import torch
|
||||
|
||||
try:
|
||||
import librosa
|
||||
except Exception: # pragma: no cover
|
||||
librosa = None
|
||||
|
||||
from fastapi import FastAPI
|
||||
from fastapi.responses import Response
|
||||
from pydantic import BaseModel
|
||||
|
||||
from cosyvoice.cli.cosyvoice import CosyVoice
|
||||
|
||||
MODEL_DIR = os.environ.get(
|
||||
"COSYVOICE_MODEL_DIR", "/models/cosyvoice/CosyVoice-300M-SFT"
|
||||
)
|
||||
DEFAULT_SPK = "韩语女" # 한국어 여성 프리셋
|
||||
|
||||
_model = None
|
||||
|
||||
|
||||
def get_model() -> CosyVoice:
|
||||
global _model
|
||||
if _model is None:
|
||||
_model = CosyVoice(
|
||||
MODEL_DIR, load_jit=False, load_trt=False, fp16=False
|
||||
)
|
||||
return _model
|
||||
|
||||
|
||||
app = FastAPI(title="CosyVoice Worker")
|
||||
|
||||
|
||||
class SynthReq(BaseModel):
|
||||
text: str
|
||||
speaker: str = DEFAULT_SPK
|
||||
speed: float = 1.0
|
||||
pitch: float = 0.0
|
||||
|
||||
|
||||
@app.on_event("startup")
|
||||
def _startup() -> None:
|
||||
get_model()
|
||||
|
||||
|
||||
@app.get("/health")
|
||||
def health() -> dict:
|
||||
m = get_model()
|
||||
return {
|
||||
"status": "ok",
|
||||
"speakers": m.list_available_spks(),
|
||||
"sr": m.sample_rate,
|
||||
"cuda": torch.cuda.is_available(),
|
||||
}
|
||||
|
||||
|
||||
@app.post("/synth")
|
||||
def synth(req: SynthReq) -> Response:
|
||||
text = (req.text or "").strip()
|
||||
if not text:
|
||||
return Response(content="empty text", status_code=400)
|
||||
|
||||
m = get_model()
|
||||
spks = m.list_available_spks()
|
||||
spk = req.speaker if req.speaker in spks else (
|
||||
DEFAULT_SPK if DEFAULT_SPK in spks else spks[0]
|
||||
)
|
||||
speed = float(max(0.5, min(2.0, req.speed)))
|
||||
pitch = float(max(-12.0, min(12.0, req.pitch)))
|
||||
|
||||
outs = list(m.inference_sft(text, spk, stream=False, speed=speed))
|
||||
audio = (
|
||||
torch.cat([o["tts_speech"] for o in outs], dim=1)
|
||||
.squeeze(0)
|
||||
.cpu()
|
||||
.numpy()
|
||||
.astype(np.float32)
|
||||
)
|
||||
sr = m.sample_rate
|
||||
|
||||
if abs(pitch) > 1e-3 and librosa is not None:
|
||||
audio = librosa.effects.pitch_shift(audio, sr=sr, n_steps=pitch)
|
||||
|
||||
buf = io.BytesIO()
|
||||
sf.write(buf, audio, sr, format="WAV", subtype="PCM_16")
|
||||
buf.seek(0)
|
||||
return Response(content=buf.read(), media_type="audio/wav")
|
||||
Reference in New Issue
Block a user