fix: .gitignore 가 소스 패키지 src/livesub/models 를 통째로 먹고 있었다
Some checks failed
Windows / verify (push) Has been cancelled
Some checks failed
Windows / verify (push) Has been cancelled
`models/` 처럼 슬래시 없이 쓴 규칙은 깊이에 상관없이 같은 이름의 폴더를 전부 무시한다. 의도는 루트의 모델 캐시였는데 소스 패키지까지 같이 걸렸고, 그래서 asr/translator/manager/tiers/glossary/packs/speech_level 8개 파일이 한 번도 커밋된 적이 없다. 증상: Gitea 에서 clone 하면 `ModuleNotFoundError: No module named 'livesub.models'` 로 앱이 아예 임포트되지 않는다. 로컬 작업 트리에는 파일이 있으니 개발 중에는 절대 안 보인다. 방금 붙인 Windows CI VM 이 첫 검증에서 바로 잡아냈다. /models/, /adapters/, /data/ 로 루트에 고정하고 빠진 파일을 추가한다.
This commit is contained in:
95
src/livesub/models/asr.py
Normal file
95
src/livesub/models/asr.py
Normal file
@@ -0,0 +1,95 @@
|
||||
"""음성인식(ASR) 래퍼 — faster-whisper(CTranslate2) 기반."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import threading
|
||||
from dataclasses import dataclass
|
||||
|
||||
import numpy as np
|
||||
|
||||
from ..constants import models_dir
|
||||
from .tiers import AsrSpec
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Transcript:
|
||||
text: str
|
||||
language: str
|
||||
confidence: float = 0.0
|
||||
duration_s: float = 0.0
|
||||
|
||||
|
||||
class SpeechRecognizer:
|
||||
"""faster-whisper 모델 한 개를 감싼다. 스레드 안전."""
|
||||
|
||||
def __init__(self, spec: AsrSpec, device: str = "cuda", device_index: int = 0) -> None:
|
||||
self.spec = spec
|
||||
self.device = device
|
||||
self.device_index = device_index
|
||||
self._model = None
|
||||
self._lock = threading.Lock()
|
||||
|
||||
@property
|
||||
def loaded(self) -> bool:
|
||||
return self._model is not None
|
||||
|
||||
def load(self) -> None:
|
||||
if self._model is not None:
|
||||
return
|
||||
from faster_whisper import WhisperModel
|
||||
|
||||
compute_type = self.spec.compute_type
|
||||
if self.device == "cpu" and compute_type.endswith("float16"):
|
||||
compute_type = "int8" # CPU에서는 float16이 오히려 느리다
|
||||
log.info("음성인식 모델 로드: %s (%s/%s)", self.spec.repo, self.device, compute_type)
|
||||
self._model = WhisperModel(
|
||||
self.spec.repo,
|
||||
device=self.device,
|
||||
device_index=self.device_index,
|
||||
compute_type=compute_type,
|
||||
download_root=str(models_dir()),
|
||||
)
|
||||
|
||||
def unload(self) -> None:
|
||||
with self._lock:
|
||||
self._model = None
|
||||
|
||||
def transcribe(
|
||||
self,
|
||||
audio: np.ndarray,
|
||||
language: str | None = None,
|
||||
fast: bool = False,
|
||||
prompt: str = "",
|
||||
) -> Transcript:
|
||||
"""오디오 한 덩어리를 텍스트로.
|
||||
|
||||
`fast=True` 는 중간 결과용으로 beam search를 끄고 빠르게 돌린다.
|
||||
"""
|
||||
if self._model is None:
|
||||
self.load()
|
||||
assert self._model is not None
|
||||
|
||||
with self._lock:
|
||||
segments, info = self._model.transcribe(
|
||||
audio.astype(np.float32, copy=False),
|
||||
language=language,
|
||||
beam_size=1 if fast else self.spec.beam_size,
|
||||
temperature=0.0,
|
||||
condition_on_previous_text=False,
|
||||
initial_prompt=prompt or None,
|
||||
vad_filter=not fast,
|
||||
vad_parameters={"min_silence_duration_ms": 300},
|
||||
word_timestamps=False,
|
||||
)
|
||||
parts = [s.text for s in segments]
|
||||
|
||||
text = " ".join(p.strip() for p in parts if p.strip()).strip()
|
||||
return Transcript(
|
||||
text=text,
|
||||
language=getattr(info, "language", language or "") or "",
|
||||
confidence=float(getattr(info, "language_probability", 0.0) or 0.0),
|
||||
duration_s=float(getattr(info, "duration", 0.0) or 0.0),
|
||||
)
|
||||
Reference in New Issue
Block a user