Some checks failed
Windows / verify (push) Has been cancelled
`models/` 처럼 슬래시 없이 쓴 규칙은 깊이에 상관없이 같은 이름의 폴더를 전부 무시한다. 의도는 루트의 모델 캐시였는데 소스 패키지까지 같이 걸렸고, 그래서 asr/translator/manager/tiers/glossary/packs/speech_level 8개 파일이 한 번도 커밋된 적이 없다. 증상: Gitea 에서 clone 하면 `ModuleNotFoundError: No module named 'livesub.models'` 로 앱이 아예 임포트되지 않는다. 로컬 작업 트리에는 파일이 있으니 개발 중에는 절대 안 보인다. 방금 붙인 Windows CI VM 이 첫 검증에서 바로 잡아냈다. /models/, /adapters/, /data/ 로 루트에 고정하고 빠진 파일을 추가한다.
96 lines
2.9 KiB
Python
96 lines
2.9 KiB
Python
"""음성인식(ASR) 래퍼 — faster-whisper(CTranslate2) 기반."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import threading
|
|
from dataclasses import dataclass
|
|
|
|
import numpy as np
|
|
|
|
from ..constants import models_dir
|
|
from .tiers import AsrSpec
|
|
|
|
log = logging.getLogger(__name__)
|
|
|
|
|
|
@dataclass
|
|
class Transcript:
|
|
text: str
|
|
language: str
|
|
confidence: float = 0.0
|
|
duration_s: float = 0.0
|
|
|
|
|
|
class SpeechRecognizer:
|
|
"""faster-whisper 모델 한 개를 감싼다. 스레드 안전."""
|
|
|
|
def __init__(self, spec: AsrSpec, device: str = "cuda", device_index: int = 0) -> None:
|
|
self.spec = spec
|
|
self.device = device
|
|
self.device_index = device_index
|
|
self._model = None
|
|
self._lock = threading.Lock()
|
|
|
|
@property
|
|
def loaded(self) -> bool:
|
|
return self._model is not None
|
|
|
|
def load(self) -> None:
|
|
if self._model is not None:
|
|
return
|
|
from faster_whisper import WhisperModel
|
|
|
|
compute_type = self.spec.compute_type
|
|
if self.device == "cpu" and compute_type.endswith("float16"):
|
|
compute_type = "int8" # CPU에서는 float16이 오히려 느리다
|
|
log.info("음성인식 모델 로드: %s (%s/%s)", self.spec.repo, self.device, compute_type)
|
|
self._model = WhisperModel(
|
|
self.spec.repo,
|
|
device=self.device,
|
|
device_index=self.device_index,
|
|
compute_type=compute_type,
|
|
download_root=str(models_dir()),
|
|
)
|
|
|
|
def unload(self) -> None:
|
|
with self._lock:
|
|
self._model = None
|
|
|
|
def transcribe(
|
|
self,
|
|
audio: np.ndarray,
|
|
language: str | None = None,
|
|
fast: bool = False,
|
|
prompt: str = "",
|
|
) -> Transcript:
|
|
"""오디오 한 덩어리를 텍스트로.
|
|
|
|
`fast=True` 는 중간 결과용으로 beam search를 끄고 빠르게 돌린다.
|
|
"""
|
|
if self._model is None:
|
|
self.load()
|
|
assert self._model is not None
|
|
|
|
with self._lock:
|
|
segments, info = self._model.transcribe(
|
|
audio.astype(np.float32, copy=False),
|
|
language=language,
|
|
beam_size=1 if fast else self.spec.beam_size,
|
|
temperature=0.0,
|
|
condition_on_previous_text=False,
|
|
initial_prompt=prompt or None,
|
|
vad_filter=not fast,
|
|
vad_parameters={"min_silence_duration_ms": 300},
|
|
word_timestamps=False,
|
|
)
|
|
parts = [s.text for s in segments]
|
|
|
|
text = " ".join(p.strip() for p in parts if p.strip()).strip()
|
|
return Transcript(
|
|
text=text,
|
|
language=getattr(info, "language", language or "") or "",
|
|
confidence=float(getattr(info, "language_probability", 0.0) or 0.0),
|
|
duration_s=float(getattr(info, "duration", 0.0) or 0.0),
|
|
)
|