Fix clone API: drop no-op instruct, serialize generation, clearer errors
- Remove the 'instruct' style field: Qwen3-TTS Base generate_voice_clone silently ignores it, so the UI option did nothing. - Serialize generations with a lock so concurrent requests queue instead of doubling RAM/VRAM on one shared model. - Validate empty text/ref_text ourselves so the API returns the Korean 400 messages instead of FastAPI's generic 422. - UI: explain that mic recording needs https/localhost instead of a raw TypeError on plain-http LAN access. - README: document mic/https limit, queueing, and measured CPU speed. Verified on CPU (Python 3.12, torch 2.14 CPU, 0.6B Base): wav and mp3 references, language ko/auto, two concurrent requests -> all 200 and Whisper transcripts match the requested text. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
37
engine.py
37
engine.py
@@ -69,6 +69,9 @@ class VoiceCloner:
|
||||
self.model_size = model_size
|
||||
self.device = device or pick_device()
|
||||
self._lock = threading.Lock()
|
||||
# One generation at a time: the model is shared and a second concurrent
|
||||
# run would just double VRAM/RAM use (and can OOM a small GPU).
|
||||
self._gen_lock = threading.Lock()
|
||||
|
||||
@property
|
||||
def loaded(self) -> bool:
|
||||
@@ -107,7 +110,6 @@ class VoiceCloner:
|
||||
ref_text: str,
|
||||
text: str,
|
||||
language: str = "auto",
|
||||
instruct: Optional[str] = None,
|
||||
seed: Optional[int] = None,
|
||||
) -> Tuple[np.ndarray, int]:
|
||||
"""Synthesize `text` in the voice of `ref_audio_path`.
|
||||
@@ -117,7 +119,6 @@ class VoiceCloner:
|
||||
ref_text: Exact transcript of the reference clip.
|
||||
text: The text to speak in the cloned voice.
|
||||
language: Language code from LANGUAGE_CODE_TO_NAME ("auto" to detect).
|
||||
instruct: Optional natural-language style hint (e.g. "speak cheerfully").
|
||||
seed: Optional random seed for reproducible output.
|
||||
|
||||
Returns:
|
||||
@@ -126,24 +127,24 @@ class VoiceCloner:
|
||||
self.load()
|
||||
import torch
|
||||
|
||||
if seed is not None:
|
||||
torch.manual_seed(seed)
|
||||
if self.device == "cuda":
|
||||
torch.cuda.manual_seed_all(seed)
|
||||
with self._gen_lock:
|
||||
if seed is not None:
|
||||
torch.manual_seed(seed)
|
||||
if self.device == "cuda":
|
||||
torch.cuda.manual_seed_all(seed)
|
||||
|
||||
# Build the speaker prompt from the reference clip + its transcript.
|
||||
voice_prompt = self.model.create_voice_clone_prompt(
|
||||
ref_audio=str(ref_audio_path),
|
||||
ref_text=ref_text,
|
||||
x_vector_only_mode=False,
|
||||
)
|
||||
# Build the speaker prompt from the reference clip + its transcript.
|
||||
voice_prompt = self.model.create_voice_clone_prompt(
|
||||
ref_audio=str(ref_audio_path),
|
||||
ref_text=ref_text,
|
||||
x_vector_only_mode=False,
|
||||
)
|
||||
|
||||
wavs, sample_rate = self.model.generate_voice_clone(
|
||||
text=text,
|
||||
voice_clone_prompt=voice_prompt,
|
||||
language=LANGUAGE_CODE_TO_NAME.get(language, "auto"),
|
||||
instruct=instruct or None,
|
||||
)
|
||||
wavs, sample_rate = self.model.generate_voice_clone(
|
||||
text=text,
|
||||
voice_clone_prompt=voice_prompt,
|
||||
language=LANGUAGE_CODE_TO_NAME.get(language, "auto"),
|
||||
)
|
||||
|
||||
audio = wavs[0]
|
||||
if isinstance(audio, torch.Tensor):
|
||||
|
||||
Reference in New Issue
Block a user