From a34f16123fc7302e5a56fdd3d4797ecbd5552f1a Mon Sep 17 00:00:00 2001 From: voice_copy bot Date: Mon, 5 Oct 2026 04:11:39 +0900 Subject: [PATCH] Fix clone API: drop no-op instruct, serialize generation, clearer errors - Remove the 'instruct' style field: Qwen3-TTS Base generate_voice_clone silently ignores it, so the UI option did nothing. - Serialize generations with a lock so concurrent requests queue instead of doubling RAM/VRAM on one shared model. - Validate empty text/ref_text ourselves so the API returns the Korean 400 messages instead of FastAPI's generic 422. - UI: explain that mic recording needs https/localhost instead of a raw TypeError on plain-http LAN access. - README: document mic/https limit, queueing, and measured CPU speed. Verified on CPU (Python 3.12, torch 2.14 CPU, 0.6B Base): wav and mp3 references, language ko/auto, two concurrent requests -> all 200 and Whisper transcripts match the requested text. Co-Authored-By: Claude Opus 5.5 --- README.md | 7 ++++++- app.py | 7 +++---- engine.py | 37 +++++++++++++++++++------------------ static/index.html | 9 ++++----- 4 files changed, 32 insertions(+), 28 deletions(-) diff --git a/README.md b/README.md index dfddaa7..605821c 100644 --- a/README.md +++ b/README.md @@ -51,10 +51,15 @@ python app.py 2. **전사** — 그 참조 음성에서 말한 내용을 그대로 텍스트로 적습니다. (클로닝 품질에 중요) 3. **합성할 텍스트** — 그 목소리로 말하게 할 문장을 입력하고 "목소리 생성"을 누릅니다. +- 마이크 녹음은 브라우저 보안 정책상 `http://localhost` 또는 `https://` 접속에서만 동작합니다. + 다른 PC에서 `http://<서버IP>:8000`으로 접속했다면 녹음 대신 파일 업로드를 사용하세요. +- 생성 요청은 한 번에 하나씩 순서대로 처리됩니다(동시 요청 시 대기). +- 참고 속도: CPU(8코어)·0.6B 모델 기준 짧은 한 문장에 수십 초~1분대(호스트 부하에 따라 다름). GPU에서는 훨씬 빠릅니다. + ## API - `GET /api/health` — 디바이스/모델/언어 상태 -- `POST /api/clone` — multipart form: `ref_audio`(파일), `ref_text`, `text`, `language`, `instruct`(선택), `seed`(선택) → `audio/wav` 반환 +- `POST /api/clone` — multipart form: `ref_audio`(파일), `ref_text`, `text`, `language`, `seed`(선택) → `audio/wav` 반환 ```bash curl -X POST http://localhost:8000/api/clone \ diff --git a/app.py b/app.py index 2740254..f5addcf 100644 --- a/app.py +++ b/app.py @@ -94,10 +94,9 @@ def health() -> dict: @app.post("/api/clone") async def clone( ref_audio: UploadFile = File(...), - ref_text: str = Form(...), - text: str = Form(...), + ref_text: str = Form(""), + text: str = Form(""), language: str = Form("auto"), - instruct: str = Form(""), seed: str = Form(""), ) -> Response: if not text.strip(): @@ -127,7 +126,7 @@ async def clone( try: audio, sr = await run_in_threadpool( - cloner.clone, wav_path, ref_text, text, language, instruct or None, seed_int + cloner.clone, wav_path, ref_text.strip(), text.strip(), language, seed_int ) except Exception as exc: # noqa: BLE001 logger.exception("clone failed") diff --git a/engine.py b/engine.py index d5d7b0e..a93c000 100644 --- a/engine.py +++ b/engine.py @@ -69,6 +69,9 @@ class VoiceCloner: self.model_size = model_size self.device = device or pick_device() self._lock = threading.Lock() + # One generation at a time: the model is shared and a second concurrent + # run would just double VRAM/RAM use (and can OOM a small GPU). + self._gen_lock = threading.Lock() @property def loaded(self) -> bool: @@ -107,7 +110,6 @@ class VoiceCloner: ref_text: str, text: str, language: str = "auto", - instruct: Optional[str] = None, seed: Optional[int] = None, ) -> Tuple[np.ndarray, int]: """Synthesize `text` in the voice of `ref_audio_path`. @@ -117,7 +119,6 @@ class VoiceCloner: ref_text: Exact transcript of the reference clip. text: The text to speak in the cloned voice. language: Language code from LANGUAGE_CODE_TO_NAME ("auto" to detect). - instruct: Optional natural-language style hint (e.g. "speak cheerfully"). seed: Optional random seed for reproducible output. Returns: @@ -126,24 +127,24 @@ class VoiceCloner: self.load() import torch - if seed is not None: - torch.manual_seed(seed) - if self.device == "cuda": - torch.cuda.manual_seed_all(seed) + with self._gen_lock: + if seed is not None: + torch.manual_seed(seed) + if self.device == "cuda": + torch.cuda.manual_seed_all(seed) - # Build the speaker prompt from the reference clip + its transcript. - voice_prompt = self.model.create_voice_clone_prompt( - ref_audio=str(ref_audio_path), - ref_text=ref_text, - x_vector_only_mode=False, - ) + # Build the speaker prompt from the reference clip + its transcript. + voice_prompt = self.model.create_voice_clone_prompt( + ref_audio=str(ref_audio_path), + ref_text=ref_text, + x_vector_only_mode=False, + ) - wavs, sample_rate = self.model.generate_voice_clone( - text=text, - voice_clone_prompt=voice_prompt, - language=LANGUAGE_CODE_TO_NAME.get(language, "auto"), - instruct=instruct or None, - ) + wavs, sample_rate = self.model.generate_voice_clone( + text=text, + voice_clone_prompt=voice_prompt, + language=LANGUAGE_CODE_TO_NAME.get(language, "auto"), + ) audio = wavs[0] if isinstance(audio, torch.Tensor): diff --git a/static/index.html b/static/index.html index 4e892c6..13cbfab 100644 --- a/static/index.html +++ b/static/index.html @@ -66,10 +66,6 @@ -
- - -
@@ -119,6 +115,10 @@ $("file").addEventListener("change", () => { // ── Mic recording → in-browser WAV (no ffmpeg needed) ── $("recBtn").addEventListener("click", async () => { + if (!window.isSecureContext || !navigator.mediaDevices) { + setStatus("브라우저 보안 정책상 마이크 녹음은 https 또는 localhost 접속에서만 됩니다. 녹음 파일을 업로드해 주세요.", "err"); + return; + } try { stream = await navigator.mediaDevices.getUserMedia({ audio: true }); } catch (e) { setStatus("마이크 접근 실패: " + e.message, "err"); return; } @@ -179,7 +179,6 @@ $("go").addEventListener("click", async () => { fd.append("ref_text", refText); fd.append("text", text); fd.append("language", $("language").value); - fd.append("instruct", $("instruct").value.trim()); fd.append("seed", $("seed").value.trim()); $("go").disabled = true;