Fix clone API: drop no-op instruct, serialize generation, clearer errors
- Remove the 'instruct' style field: Qwen3-TTS Base generate_voice_clone silently ignores it, so the UI option did nothing. - Serialize generations with a lock so concurrent requests queue instead of doubling RAM/VRAM on one shared model. - Validate empty text/ref_text ourselves so the API returns the Korean 400 messages instead of FastAPI's generic 422. - UI: explain that mic recording needs https/localhost instead of a raw TypeError on plain-http LAN access. - README: document mic/https limit, queueing, and measured CPU speed. Verified on CPU (Python 3.12, torch 2.14 CPU, 0.6B Base): wav and mp3 references, language ko/auto, two concurrent requests -> all 200 and Whisper transcripts match the requested text. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -51,10 +51,15 @@ python app.py
|
|||||||
2. **전사** — 그 참조 음성에서 말한 내용을 그대로 텍스트로 적습니다. (클로닝 품질에 중요)
|
2. **전사** — 그 참조 음성에서 말한 내용을 그대로 텍스트로 적습니다. (클로닝 품질에 중요)
|
||||||
3. **합성할 텍스트** — 그 목소리로 말하게 할 문장을 입력하고 "목소리 생성"을 누릅니다.
|
3. **합성할 텍스트** — 그 목소리로 말하게 할 문장을 입력하고 "목소리 생성"을 누릅니다.
|
||||||
|
|
||||||
|
- 마이크 녹음은 브라우저 보안 정책상 `http://localhost` 또는 `https://` 접속에서만 동작합니다.
|
||||||
|
다른 PC에서 `http://<서버IP>:8000`으로 접속했다면 녹음 대신 파일 업로드를 사용하세요.
|
||||||
|
- 생성 요청은 한 번에 하나씩 순서대로 처리됩니다(동시 요청 시 대기).
|
||||||
|
- 참고 속도: CPU(8코어)·0.6B 모델 기준 짧은 한 문장에 수십 초~1분대(호스트 부하에 따라 다름). GPU에서는 훨씬 빠릅니다.
|
||||||
|
|
||||||
## API
|
## API
|
||||||
|
|
||||||
- `GET /api/health` — 디바이스/모델/언어 상태
|
- `GET /api/health` — 디바이스/모델/언어 상태
|
||||||
- `POST /api/clone` — multipart form: `ref_audio`(파일), `ref_text`, `text`, `language`, `instruct`(선택), `seed`(선택) → `audio/wav` 반환
|
- `POST /api/clone` — multipart form: `ref_audio`(파일), `ref_text`, `text`, `language`, `seed`(선택) → `audio/wav` 반환
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
curl -X POST http://localhost:8000/api/clone \
|
curl -X POST http://localhost:8000/api/clone \
|
||||||
|
|||||||
7
app.py
7
app.py
@@ -94,10 +94,9 @@ def health() -> dict:
|
|||||||
@app.post("/api/clone")
|
@app.post("/api/clone")
|
||||||
async def clone(
|
async def clone(
|
||||||
ref_audio: UploadFile = File(...),
|
ref_audio: UploadFile = File(...),
|
||||||
ref_text: str = Form(...),
|
ref_text: str = Form(""),
|
||||||
text: str = Form(...),
|
text: str = Form(""),
|
||||||
language: str = Form("auto"),
|
language: str = Form("auto"),
|
||||||
instruct: str = Form(""),
|
|
||||||
seed: str = Form(""),
|
seed: str = Form(""),
|
||||||
) -> Response:
|
) -> Response:
|
||||||
if not text.strip():
|
if not text.strip():
|
||||||
@@ -127,7 +126,7 @@ async def clone(
|
|||||||
|
|
||||||
try:
|
try:
|
||||||
audio, sr = await run_in_threadpool(
|
audio, sr = await run_in_threadpool(
|
||||||
cloner.clone, wav_path, ref_text, text, language, instruct or None, seed_int
|
cloner.clone, wav_path, ref_text.strip(), text.strip(), language, seed_int
|
||||||
)
|
)
|
||||||
except Exception as exc: # noqa: BLE001
|
except Exception as exc: # noqa: BLE001
|
||||||
logger.exception("clone failed")
|
logger.exception("clone failed")
|
||||||
|
|||||||
37
engine.py
37
engine.py
@@ -69,6 +69,9 @@ class VoiceCloner:
|
|||||||
self.model_size = model_size
|
self.model_size = model_size
|
||||||
self.device = device or pick_device()
|
self.device = device or pick_device()
|
||||||
self._lock = threading.Lock()
|
self._lock = threading.Lock()
|
||||||
|
# One generation at a time: the model is shared and a second concurrent
|
||||||
|
# run would just double VRAM/RAM use (and can OOM a small GPU).
|
||||||
|
self._gen_lock = threading.Lock()
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def loaded(self) -> bool:
|
def loaded(self) -> bool:
|
||||||
@@ -107,7 +110,6 @@ class VoiceCloner:
|
|||||||
ref_text: str,
|
ref_text: str,
|
||||||
text: str,
|
text: str,
|
||||||
language: str = "auto",
|
language: str = "auto",
|
||||||
instruct: Optional[str] = None,
|
|
||||||
seed: Optional[int] = None,
|
seed: Optional[int] = None,
|
||||||
) -> Tuple[np.ndarray, int]:
|
) -> Tuple[np.ndarray, int]:
|
||||||
"""Synthesize `text` in the voice of `ref_audio_path`.
|
"""Synthesize `text` in the voice of `ref_audio_path`.
|
||||||
@@ -117,7 +119,6 @@ class VoiceCloner:
|
|||||||
ref_text: Exact transcript of the reference clip.
|
ref_text: Exact transcript of the reference clip.
|
||||||
text: The text to speak in the cloned voice.
|
text: The text to speak in the cloned voice.
|
||||||
language: Language code from LANGUAGE_CODE_TO_NAME ("auto" to detect).
|
language: Language code from LANGUAGE_CODE_TO_NAME ("auto" to detect).
|
||||||
instruct: Optional natural-language style hint (e.g. "speak cheerfully").
|
|
||||||
seed: Optional random seed for reproducible output.
|
seed: Optional random seed for reproducible output.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
@@ -126,24 +127,24 @@ class VoiceCloner:
|
|||||||
self.load()
|
self.load()
|
||||||
import torch
|
import torch
|
||||||
|
|
||||||
if seed is not None:
|
with self._gen_lock:
|
||||||
torch.manual_seed(seed)
|
if seed is not None:
|
||||||
if self.device == "cuda":
|
torch.manual_seed(seed)
|
||||||
torch.cuda.manual_seed_all(seed)
|
if self.device == "cuda":
|
||||||
|
torch.cuda.manual_seed_all(seed)
|
||||||
|
|
||||||
# Build the speaker prompt from the reference clip + its transcript.
|
# Build the speaker prompt from the reference clip + its transcript.
|
||||||
voice_prompt = self.model.create_voice_clone_prompt(
|
voice_prompt = self.model.create_voice_clone_prompt(
|
||||||
ref_audio=str(ref_audio_path),
|
ref_audio=str(ref_audio_path),
|
||||||
ref_text=ref_text,
|
ref_text=ref_text,
|
||||||
x_vector_only_mode=False,
|
x_vector_only_mode=False,
|
||||||
)
|
)
|
||||||
|
|
||||||
wavs, sample_rate = self.model.generate_voice_clone(
|
wavs, sample_rate = self.model.generate_voice_clone(
|
||||||
text=text,
|
text=text,
|
||||||
voice_clone_prompt=voice_prompt,
|
voice_clone_prompt=voice_prompt,
|
||||||
language=LANGUAGE_CODE_TO_NAME.get(language, "auto"),
|
language=LANGUAGE_CODE_TO_NAME.get(language, "auto"),
|
||||||
instruct=instruct or None,
|
)
|
||||||
)
|
|
||||||
|
|
||||||
audio = wavs[0]
|
audio = wavs[0]
|
||||||
if isinstance(audio, torch.Tensor):
|
if isinstance(audio, torch.Tensor):
|
||||||
|
|||||||
@@ -66,10 +66,6 @@
|
|||||||
<label>언어</label>
|
<label>언어</label>
|
||||||
<select id="language"></select>
|
<select id="language"></select>
|
||||||
</div>
|
</div>
|
||||||
<div>
|
|
||||||
<label>스타일 지시 <span class="hint">(선택)</span></label>
|
|
||||||
<input type="text" id="instruct" placeholder="예) 밝고 활기차게" />
|
|
||||||
</div>
|
|
||||||
<div>
|
<div>
|
||||||
<label>seed <span class="hint">(선택)</span></label>
|
<label>seed <span class="hint">(선택)</span></label>
|
||||||
<input type="number" id="seed" placeholder="랜덤" />
|
<input type="number" id="seed" placeholder="랜덤" />
|
||||||
@@ -119,6 +115,10 @@ $("file").addEventListener("change", () => {
|
|||||||
|
|
||||||
// ── Mic recording → in-browser WAV (no ffmpeg needed) ──
|
// ── Mic recording → in-browser WAV (no ffmpeg needed) ──
|
||||||
$("recBtn").addEventListener("click", async () => {
|
$("recBtn").addEventListener("click", async () => {
|
||||||
|
if (!window.isSecureContext || !navigator.mediaDevices) {
|
||||||
|
setStatus("브라우저 보안 정책상 마이크 녹음은 https 또는 localhost 접속에서만 됩니다. 녹음 파일을 업로드해 주세요.", "err");
|
||||||
|
return;
|
||||||
|
}
|
||||||
try {
|
try {
|
||||||
stream = await navigator.mediaDevices.getUserMedia({ audio: true });
|
stream = await navigator.mediaDevices.getUserMedia({ audio: true });
|
||||||
} catch (e) { setStatus("마이크 접근 실패: " + e.message, "err"); return; }
|
} catch (e) { setStatus("마이크 접근 실패: " + e.message, "err"); return; }
|
||||||
@@ -179,7 +179,6 @@ $("go").addEventListener("click", async () => {
|
|||||||
fd.append("ref_text", refText);
|
fd.append("ref_text", refText);
|
||||||
fd.append("text", text);
|
fd.append("text", text);
|
||||||
fd.append("language", $("language").value);
|
fd.append("language", $("language").value);
|
||||||
fd.append("instruct", $("instruct").value.trim());
|
|
||||||
fd.append("seed", $("seed").value.trim());
|
fd.append("seed", $("seed").value.trim());
|
||||||
|
|
||||||
$("go").disabled = true;
|
$("go").disabled = true;
|
||||||
|
|||||||
Reference in New Issue
Block a user