Fix clone API: drop no-op instruct, serialize generation, clearer errors
- Remove the 'instruct' style field: Qwen3-TTS Base generate_voice_clone silently ignores it, so the UI option did nothing. - Serialize generations with a lock so concurrent requests queue instead of doubling RAM/VRAM on one shared model. - Validate empty text/ref_text ourselves so the API returns the Korean 400 messages instead of FastAPI's generic 422. - UI: explain that mic recording needs https/localhost instead of a raw TypeError on plain-http LAN access. - README: document mic/https limit, queueing, and measured CPU speed. Verified on CPU (Python 3.12, torch 2.14 CPU, 0.6B Base): wav and mp3 references, language ko/auto, two concurrent requests -> all 200 and Whisper transcripts match the requested text. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -51,10 +51,15 @@ python app.py
|
||||
2. **전사** — 그 참조 음성에서 말한 내용을 그대로 텍스트로 적습니다. (클로닝 품질에 중요)
|
||||
3. **합성할 텍스트** — 그 목소리로 말하게 할 문장을 입력하고 "목소리 생성"을 누릅니다.
|
||||
|
||||
- 마이크 녹음은 브라우저 보안 정책상 `http://localhost` 또는 `https://` 접속에서만 동작합니다.
|
||||
다른 PC에서 `http://<서버IP>:8000`으로 접속했다면 녹음 대신 파일 업로드를 사용하세요.
|
||||
- 생성 요청은 한 번에 하나씩 순서대로 처리됩니다(동시 요청 시 대기).
|
||||
- 참고 속도: CPU(8코어)·0.6B 모델 기준 짧은 한 문장에 수십 초~1분대(호스트 부하에 따라 다름). GPU에서는 훨씬 빠릅니다.
|
||||
|
||||
## API
|
||||
|
||||
- `GET /api/health` — 디바이스/모델/언어 상태
|
||||
- `POST /api/clone` — multipart form: `ref_audio`(파일), `ref_text`, `text`, `language`, `instruct`(선택), `seed`(선택) → `audio/wav` 반환
|
||||
- `POST /api/clone` — multipart form: `ref_audio`(파일), `ref_text`, `text`, `language`, `seed`(선택) → `audio/wav` 반환
|
||||
|
||||
```bash
|
||||
curl -X POST http://localhost:8000/api/clone \
|
||||
|
||||
7
app.py
7
app.py
@@ -94,10 +94,9 @@ def health() -> dict:
|
||||
@app.post("/api/clone")
|
||||
async def clone(
|
||||
ref_audio: UploadFile = File(...),
|
||||
ref_text: str = Form(...),
|
||||
text: str = Form(...),
|
||||
ref_text: str = Form(""),
|
||||
text: str = Form(""),
|
||||
language: str = Form("auto"),
|
||||
instruct: str = Form(""),
|
||||
seed: str = Form(""),
|
||||
) -> Response:
|
||||
if not text.strip():
|
||||
@@ -127,7 +126,7 @@ async def clone(
|
||||
|
||||
try:
|
||||
audio, sr = await run_in_threadpool(
|
||||
cloner.clone, wav_path, ref_text, text, language, instruct or None, seed_int
|
||||
cloner.clone, wav_path, ref_text.strip(), text.strip(), language, seed_int
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.exception("clone failed")
|
||||
|
||||
37
engine.py
37
engine.py
@@ -69,6 +69,9 @@ class VoiceCloner:
|
||||
self.model_size = model_size
|
||||
self.device = device or pick_device()
|
||||
self._lock = threading.Lock()
|
||||
# One generation at a time: the model is shared and a second concurrent
|
||||
# run would just double VRAM/RAM use (and can OOM a small GPU).
|
||||
self._gen_lock = threading.Lock()
|
||||
|
||||
@property
|
||||
def loaded(self) -> bool:
|
||||
@@ -107,7 +110,6 @@ class VoiceCloner:
|
||||
ref_text: str,
|
||||
text: str,
|
||||
language: str = "auto",
|
||||
instruct: Optional[str] = None,
|
||||
seed: Optional[int] = None,
|
||||
) -> Tuple[np.ndarray, int]:
|
||||
"""Synthesize `text` in the voice of `ref_audio_path`.
|
||||
@@ -117,7 +119,6 @@ class VoiceCloner:
|
||||
ref_text: Exact transcript of the reference clip.
|
||||
text: The text to speak in the cloned voice.
|
||||
language: Language code from LANGUAGE_CODE_TO_NAME ("auto" to detect).
|
||||
instruct: Optional natural-language style hint (e.g. "speak cheerfully").
|
||||
seed: Optional random seed for reproducible output.
|
||||
|
||||
Returns:
|
||||
@@ -126,24 +127,24 @@ class VoiceCloner:
|
||||
self.load()
|
||||
import torch
|
||||
|
||||
if seed is not None:
|
||||
torch.manual_seed(seed)
|
||||
if self.device == "cuda":
|
||||
torch.cuda.manual_seed_all(seed)
|
||||
with self._gen_lock:
|
||||
if seed is not None:
|
||||
torch.manual_seed(seed)
|
||||
if self.device == "cuda":
|
||||
torch.cuda.manual_seed_all(seed)
|
||||
|
||||
# Build the speaker prompt from the reference clip + its transcript.
|
||||
voice_prompt = self.model.create_voice_clone_prompt(
|
||||
ref_audio=str(ref_audio_path),
|
||||
ref_text=ref_text,
|
||||
x_vector_only_mode=False,
|
||||
)
|
||||
# Build the speaker prompt from the reference clip + its transcript.
|
||||
voice_prompt = self.model.create_voice_clone_prompt(
|
||||
ref_audio=str(ref_audio_path),
|
||||
ref_text=ref_text,
|
||||
x_vector_only_mode=False,
|
||||
)
|
||||
|
||||
wavs, sample_rate = self.model.generate_voice_clone(
|
||||
text=text,
|
||||
voice_clone_prompt=voice_prompt,
|
||||
language=LANGUAGE_CODE_TO_NAME.get(language, "auto"),
|
||||
instruct=instruct or None,
|
||||
)
|
||||
wavs, sample_rate = self.model.generate_voice_clone(
|
||||
text=text,
|
||||
voice_clone_prompt=voice_prompt,
|
||||
language=LANGUAGE_CODE_TO_NAME.get(language, "auto"),
|
||||
)
|
||||
|
||||
audio = wavs[0]
|
||||
if isinstance(audio, torch.Tensor):
|
||||
|
||||
@@ -66,10 +66,6 @@
|
||||
<label>언어</label>
|
||||
<select id="language"></select>
|
||||
</div>
|
||||
<div>
|
||||
<label>스타일 지시 <span class="hint">(선택)</span></label>
|
||||
<input type="text" id="instruct" placeholder="예) 밝고 활기차게" />
|
||||
</div>
|
||||
<div>
|
||||
<label>seed <span class="hint">(선택)</span></label>
|
||||
<input type="number" id="seed" placeholder="랜덤" />
|
||||
@@ -119,6 +115,10 @@ $("file").addEventListener("change", () => {
|
||||
|
||||
// ── Mic recording → in-browser WAV (no ffmpeg needed) ──
|
||||
$("recBtn").addEventListener("click", async () => {
|
||||
if (!window.isSecureContext || !navigator.mediaDevices) {
|
||||
setStatus("브라우저 보안 정책상 마이크 녹음은 https 또는 localhost 접속에서만 됩니다. 녹음 파일을 업로드해 주세요.", "err");
|
||||
return;
|
||||
}
|
||||
try {
|
||||
stream = await navigator.mediaDevices.getUserMedia({ audio: true });
|
||||
} catch (e) { setStatus("마이크 접근 실패: " + e.message, "err"); return; }
|
||||
@@ -179,7 +179,6 @@ $("go").addEventListener("click", async () => {
|
||||
fd.append("ref_text", refText);
|
||||
fd.append("text", text);
|
||||
fd.append("language", $("language").value);
|
||||
fd.append("instruct", $("instruct").value.trim());
|
||||
fd.append("seed", $("seed").value.trim());
|
||||
|
||||
$("go").disabled = true;
|
||||
|
||||
Reference in New Issue
Block a user