이름을 Hearo → LiveSub 로 변경. 말장난보다 하는 일이 바로 보이는 쪽이 낫다. 자막 on/off - 전역 단축키 Ctrl+Alt+S (자막) / Ctrl+Alt+D (번역) — Windows RegisterHotKey + 네이티브 이벤트 필터라 게임 창이 떠 있어도 동작. 다른 OS 에서는 no-op - 단축키·체크박스·우클릭 메뉴가 모두 같은 경로를 타도록 통합 모니터 선택 + 디스코드식 배치 - placement.py: Qt 비의존 배치 계산 (모니터 목록 → 9분할 좌표) - AnchorGrid 위젯으로 3x3 위치 선택, 드래그하면 자유 배치로 전환 - 모니터 구성이 바뀌어도 자막이 화면 밖으로 사라지지 않도록 클램프 GPU 저부하 모드 (기본 켜짐) - 연산 정밀도 int8 강등, VRAM 상한 35%, 추론 후 60ms 양보, 300초 무음 시 모델 언로드, 중간 결과 비활성 - 기본 티어를 균형(6GB) → 신속(4GB) 으로 하향 - 8GB GPU 기준 자막 2.8GB / 게임 5.2GB 게임 용어집 번들 (304개) - 롤 74 · 발로란트 59 · 오버워치2 49 · FPS공통 46 · 마크 38 · 방송 38 - 체크박스로 켜고 끄며, 사용자가 직접 등록한 항목이 항상 우선 검증: pytest 91개 통과 (배치 14 + 팩 12 + UI 6 신규), ruff clean Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
147 lines
4.5 KiB
Python
147 lines
4.5 KiB
Python
"""오디오 캡처 백엔드 공통 인터페이스.
|
|
|
|
모든 백엔드는 16kHz / mono / float32(-1.0~1.0) 프레임을 큐로 흘려보낸다.
|
|
리샘플링과 다운믹스는 각 백엔드가 책임진다.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import abc
|
|
import queue
|
|
import threading
|
|
from dataclasses import dataclass
|
|
|
|
import numpy as np
|
|
|
|
from ..constants import SAMPLE_RATE
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class AudioSource:
|
|
"""UI에 노출되는 캡처 대상 한 줄."""
|
|
|
|
kind: str # "process" | "device"
|
|
identifier: str # pid 문자열 또는 장치 인덱스
|
|
label: str
|
|
detail: str = ""
|
|
icon_path: str = ""
|
|
|
|
@property
|
|
def pid(self) -> int:
|
|
return int(self.identifier) if self.kind == "process" else 0
|
|
|
|
@property
|
|
def device_index(self) -> int:
|
|
return int(self.identifier) if self.kind == "device" else -1
|
|
|
|
|
|
class CaptureError(RuntimeError):
|
|
"""캡처를 시작할 수 없을 때."""
|
|
|
|
|
|
class CaptureBackend(abc.ABC):
|
|
"""오디오 캡처 백엔드 베이스."""
|
|
|
|
name = "base"
|
|
#: 특정 프로그램 소리만 분리해서 받을 수 있는가
|
|
per_process = False
|
|
|
|
def __init__(self, max_queue_frames: int = 400) -> None:
|
|
self._queue: queue.Queue[np.ndarray] = queue.Queue(maxsize=max_queue_frames)
|
|
self._stop = threading.Event()
|
|
self._thread: threading.Thread | None = None
|
|
self._error: Exception | None = None
|
|
|
|
# --- 하위 클래스가 구현 -------------------------------------------
|
|
@abc.abstractmethod
|
|
def _run(self) -> None:
|
|
"""블로킹 캡처 루프. self._stop 이 set 될 때까지 _emit() 호출."""
|
|
|
|
@staticmethod
|
|
@abc.abstractmethod
|
|
def available() -> bool:
|
|
"""현재 환경에서 이 백엔드를 쓸 수 있는가."""
|
|
|
|
@staticmethod
|
|
@abc.abstractmethod
|
|
def list_sources() -> list[AudioSource]:
|
|
"""선택 가능한 캡처 대상 목록."""
|
|
|
|
# --- 공통 동작 -----------------------------------------------------
|
|
def start(self) -> None:
|
|
if self._thread and self._thread.is_alive():
|
|
return
|
|
self._stop.clear()
|
|
self._error = None
|
|
self._thread = threading.Thread(
|
|
target=self._thread_main, name=f"capture-{self.name}", daemon=True
|
|
)
|
|
self._thread.start()
|
|
|
|
def _thread_main(self) -> None:
|
|
try:
|
|
self._run()
|
|
except Exception as exc: # noqa: BLE001 - 워커 스레드 경계
|
|
self._error = exc
|
|
|
|
def stop(self, timeout: float = 2.0) -> None:
|
|
self._stop.set()
|
|
if self._thread:
|
|
self._thread.join(timeout=timeout)
|
|
self._thread = None
|
|
|
|
@property
|
|
def running(self) -> bool:
|
|
return bool(self._thread and self._thread.is_alive())
|
|
|
|
@property
|
|
def error(self) -> Exception | None:
|
|
return self._error
|
|
|
|
def _emit(self, frame: np.ndarray) -> None:
|
|
"""캡처 프레임 투입. 큐가 가득 차면 가장 오래된 프레임을 버린다.
|
|
|
|
실시간 자막에서는 밀린 오디오보다 최신 오디오가 항상 더 가치 있다.
|
|
"""
|
|
try:
|
|
self._queue.put_nowait(frame)
|
|
except queue.Full:
|
|
try:
|
|
self._queue.get_nowait()
|
|
self._queue.put_nowait(frame)
|
|
except (queue.Empty, queue.Full):
|
|
pass
|
|
|
|
def read(self, timeout: float = 0.5) -> np.ndarray | None:
|
|
try:
|
|
return self._queue.get(timeout=timeout)
|
|
except queue.Empty:
|
|
return None
|
|
|
|
def drain(self) -> None:
|
|
while not self._queue.empty():
|
|
try:
|
|
self._queue.get_nowait()
|
|
except queue.Empty:
|
|
break
|
|
|
|
|
|
def to_mono_16k(data: np.ndarray, channels: int, src_rate: int) -> np.ndarray:
|
|
"""인터리브된 float32 PCM을 16kHz 모노로 변환."""
|
|
if data.size == 0:
|
|
return data.astype(np.float32, copy=False)
|
|
audio = data.astype(np.float32, copy=False)
|
|
if channels > 1:
|
|
usable = (audio.size // channels) * channels
|
|
audio = audio[:usable].reshape(-1, channels).mean(axis=1)
|
|
if src_rate != SAMPLE_RATE and audio.size:
|
|
# 선형 보간 리샘플. 음성인식 입력으로는 충분한 품질이며 의존성이 없다.
|
|
duration = audio.size / src_rate
|
|
target_len = max(1, int(duration * SAMPLE_RATE))
|
|
audio = np.interp(
|
|
np.linspace(0.0, audio.size - 1, target_len, dtype=np.float64),
|
|
np.arange(audio.size, dtype=np.float64),
|
|
audio,
|
|
).astype(np.float32)
|
|
return audio
|