diff --git a/README.md b/README.md index dfddaa7..605821c 100644 --- a/README.md +++ b/README.md @@ -51,10 +51,15 @@ python app.py 2. **전사** — 그 참조 음성에서 말한 내용을 그대로 텍스트로 적습니다. (클로닝 품질에 중요) 3. **합성할 텍스트** — 그 목소리로 말하게 할 문장을 입력하고 "목소리 생성"을 누릅니다. +- 마이크 녹음은 브라우저 보안 정책상 `http://localhost` 또는 `https://` 접속에서만 동작합니다. + 다른 PC에서 `http://<서버IP>:8000`으로 접속했다면 녹음 대신 파일 업로드를 사용하세요. +- 생성 요청은 한 번에 하나씩 순서대로 처리됩니다(동시 요청 시 대기). +- 참고 속도: CPU(8코어)·0.6B 모델 기준 짧은 한 문장에 수십 초~1분대(호스트 부하에 따라 다름). GPU에서는 훨씬 빠릅니다. + ## API - `GET /api/health` — 디바이스/모델/언어 상태 -- `POST /api/clone` — multipart form: `ref_audio`(파일), `ref_text`, `text`, `language`, `instruct`(선택), `seed`(선택) → `audio/wav` 반환 +- `POST /api/clone` — multipart form: `ref_audio`(파일), `ref_text`, `text`, `language`, `seed`(선택) → `audio/wav` 반환 ```bash curl -X POST http://localhost:8000/api/clone \ diff --git a/app.py b/app.py index 2740254..f5addcf 100644 --- a/app.py +++ b/app.py @@ -94,10 +94,9 @@ def health() -> dict: @app.post("/api/clone") async def clone( ref_audio: UploadFile = File(...), - ref_text: str = Form(...), - text: str = Form(...), + ref_text: str = Form(""), + text: str = Form(""), language: str = Form("auto"), - instruct: str = Form(""), seed: str = Form(""), ) -> Response: if not text.strip(): @@ -127,7 +126,7 @@ async def clone( try: audio, sr = await run_in_threadpool( - cloner.clone, wav_path, ref_text, text, language, instruct or None, seed_int + cloner.clone, wav_path, ref_text.strip(), text.strip(), language, seed_int ) except Exception as exc: # noqa: BLE001 logger.exception("clone failed") diff --git a/engine.py b/engine.py index d5d7b0e..a93c000 100644 --- a/engine.py +++ b/engine.py @@ -69,6 +69,9 @@ class VoiceCloner: self.model_size = model_size self.device = device or pick_device() self._lock = threading.Lock() + # One generation at a time: the model is shared and a second concurrent + # run would just double VRAM/RAM use (and can OOM a small GPU). + self._gen_lock = threading.Lock() @property def loaded(self) -> bool: @@ -107,7 +110,6 @@ class VoiceCloner: ref_text: str, text: str, language: str = "auto", - instruct: Optional[str] = None, seed: Optional[int] = None, ) -> Tuple[np.ndarray, int]: """Synthesize `text` in the voice of `ref_audio_path`. @@ -117,7 +119,6 @@ class VoiceCloner: ref_text: Exact transcript of the reference clip. text: The text to speak in the cloned voice. language: Language code from LANGUAGE_CODE_TO_NAME ("auto" to detect). - instruct: Optional natural-language style hint (e.g. "speak cheerfully"). seed: Optional random seed for reproducible output. Returns: @@ -126,24 +127,24 @@ class VoiceCloner: self.load() import torch - if seed is not None: - torch.manual_seed(seed) - if self.device == "cuda": - torch.cuda.manual_seed_all(seed) + with self._gen_lock: + if seed is not None: + torch.manual_seed(seed) + if self.device == "cuda": + torch.cuda.manual_seed_all(seed) - # Build the speaker prompt from the reference clip + its transcript. - voice_prompt = self.model.create_voice_clone_prompt( - ref_audio=str(ref_audio_path), - ref_text=ref_text, - x_vector_only_mode=False, - ) + # Build the speaker prompt from the reference clip + its transcript. + voice_prompt = self.model.create_voice_clone_prompt( + ref_audio=str(ref_audio_path), + ref_text=ref_text, + x_vector_only_mode=False, + ) - wavs, sample_rate = self.model.generate_voice_clone( - text=text, - voice_clone_prompt=voice_prompt, - language=LANGUAGE_CODE_TO_NAME.get(language, "auto"), - instruct=instruct or None, - ) + wavs, sample_rate = self.model.generate_voice_clone( + text=text, + voice_clone_prompt=voice_prompt, + language=LANGUAGE_CODE_TO_NAME.get(language, "auto"), + ) audio = wavs[0] if isinstance(audio, torch.Tensor): diff --git a/static/index.html b/static/index.html index 4e892c6..13cbfab 100644 --- a/static/index.html +++ b/static/index.html @@ -66,10 +66,6 @@ -