Files
live-app-translator/src/livesub/models/asr.py
EJClaw f9fd86464d
Some checks failed
Windows / verify (push) Has been cancelled
fix: .gitignore 가 소스 패키지 src/livesub/models 를 통째로 먹고 있었다
`models/` 처럼 슬래시 없이 쓴 규칙은 깊이에 상관없이 같은 이름의 폴더를 전부
무시한다. 의도는 루트의 모델 캐시였는데 소스 패키지까지 같이 걸렸고, 그래서
asr/translator/manager/tiers/glossary/packs/speech_level 8개 파일이 한 번도
커밋된 적이 없다.

증상: Gitea 에서 clone 하면 `ModuleNotFoundError: No module named
'livesub.models'` 로 앱이 아예 임포트되지 않는다. 로컬 작업 트리에는 파일이
있으니 개발 중에는 절대 안 보인다. 방금 붙인 Windows CI VM 이 첫 검증에서
바로 잡아냈다.

/models/, /adapters/, /data/ 로 루트에 고정하고 빠진 파일을 추가한다.
2026-09-25 23:01:31 +09:00

96 lines
2.9 KiB
Python

"""음성인식(ASR) 래퍼 — faster-whisper(CTranslate2) 기반."""
from __future__ import annotations
import logging
import threading
from dataclasses import dataclass
import numpy as np
from ..constants import models_dir
from .tiers import AsrSpec
log = logging.getLogger(__name__)
@dataclass
class Transcript:
text: str
language: str
confidence: float = 0.0
duration_s: float = 0.0
class SpeechRecognizer:
"""faster-whisper 모델 한 개를 감싼다. 스레드 안전."""
def __init__(self, spec: AsrSpec, device: str = "cuda", device_index: int = 0) -> None:
self.spec = spec
self.device = device
self.device_index = device_index
self._model = None
self._lock = threading.Lock()
@property
def loaded(self) -> bool:
return self._model is not None
def load(self) -> None:
if self._model is not None:
return
from faster_whisper import WhisperModel
compute_type = self.spec.compute_type
if self.device == "cpu" and compute_type.endswith("float16"):
compute_type = "int8" # CPU에서는 float16이 오히려 느리다
log.info("음성인식 모델 로드: %s (%s/%s)", self.spec.repo, self.device, compute_type)
self._model = WhisperModel(
self.spec.repo,
device=self.device,
device_index=self.device_index,
compute_type=compute_type,
download_root=str(models_dir()),
)
def unload(self) -> None:
with self._lock:
self._model = None
def transcribe(
self,
audio: np.ndarray,
language: str | None = None,
fast: bool = False,
prompt: str = "",
) -> Transcript:
"""오디오 한 덩어리를 텍스트로.
`fast=True` 는 중간 결과용으로 beam search를 끄고 빠르게 돌린다.
"""
if self._model is None:
self.load()
assert self._model is not None
with self._lock:
segments, info = self._model.transcribe(
audio.astype(np.float32, copy=False),
language=language,
beam_size=1 if fast else self.spec.beam_size,
temperature=0.0,
condition_on_previous_text=False,
initial_prompt=prompt or None,
vad_filter=not fast,
vad_parameters={"min_silence_duration_ms": 300},
word_timestamps=False,
)
parts = [s.text for s in segments]
text = " ".join(p.strip() for p in parts if p.strip()).strip()
return Transcript(
text=text,
language=getattr(info, "language", language or "") or "",
confidence=float(getattr(info, "language_probability", 0.0) or 0.0),
duration_s=float(getattr(info, "duration", 0.0) or 0.0),
)