feat: LiveSub 2차 — 자막 on/off, 모니터·위치 선택, GPU 저부하, 게임 용어집

이름을 Hearo → LiveSub 로 변경. 말장난보다 하는 일이 바로 보이는 쪽이 낫다.

자막 on/off
- 전역 단축키 Ctrl+Alt+S (자막) / Ctrl+Alt+D (번역) — Windows RegisterHotKey +
  네이티브 이벤트 필터라 게임 창이 떠 있어도 동작. 다른 OS 에서는 no-op
- 단축키·체크박스·우클릭 메뉴가 모두 같은 경로를 타도록 통합

모니터 선택 + 디스코드식 배치
- placement.py: Qt 비의존 배치 계산 (모니터 목록 → 9분할 좌표)
- AnchorGrid 위젯으로 3x3 위치 선택, 드래그하면 자유 배치로 전환
- 모니터 구성이 바뀌어도 자막이 화면 밖으로 사라지지 않도록 클램프

GPU 저부하 모드 (기본 켜짐)
- 연산 정밀도 int8 강등, VRAM 상한 35%, 추론 후 60ms 양보,
  300초 무음 시 모델 언로드, 중간 결과 비활성
- 기본 티어를 균형(6GB) → 신속(4GB) 으로 하향
- 8GB GPU 기준 자막 2.8GB / 게임 5.2GB

게임 용어집 번들 (304개)
- 롤 74 · 발로란트 59 · 오버워치2 49 · FPS공통 46 · 마크 38 · 방송 38
- 체크박스로 켜고 끄며, 사용자가 직접 등록한 항목이 항상 우선

검증: pytest 91개 통과 (배치 14 + 팩 12 + UI 6 신규), ruff clean

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
EJClaw
2026-09-21 17:49:26 +09:00
parent 5596f905d1
commit 8b36b244ff
60 changed files with 1939 additions and 300 deletions

146
src/livesub/audio/base.py Normal file
View File

@@ -0,0 +1,146 @@
"""오디오 캡처 백엔드 공통 인터페이스.
모든 백엔드는 16kHz / mono / float32(-1.0~1.0) 프레임을 큐로 흘려보낸다.
리샘플링과 다운믹스는 각 백엔드가 책임진다.
"""
from __future__ import annotations
import abc
import queue
import threading
from dataclasses import dataclass
import numpy as np
from ..constants import SAMPLE_RATE
@dataclass(frozen=True)
class AudioSource:
"""UI에 노출되는 캡처 대상 한 줄."""
kind: str # "process" | "device"
identifier: str # pid 문자열 또는 장치 인덱스
label: str
detail: str = ""
icon_path: str = ""
@property
def pid(self) -> int:
return int(self.identifier) if self.kind == "process" else 0
@property
def device_index(self) -> int:
return int(self.identifier) if self.kind == "device" else -1
class CaptureError(RuntimeError):
"""캡처를 시작할 수 없을 때."""
class CaptureBackend(abc.ABC):
"""오디오 캡처 백엔드 베이스."""
name = "base"
#: 특정 프로그램 소리만 분리해서 받을 수 있는가
per_process = False
def __init__(self, max_queue_frames: int = 400) -> None:
self._queue: queue.Queue[np.ndarray] = queue.Queue(maxsize=max_queue_frames)
self._stop = threading.Event()
self._thread: threading.Thread | None = None
self._error: Exception | None = None
# --- 하위 클래스가 구현 -------------------------------------------
@abc.abstractmethod
def _run(self) -> None:
"""블로킹 캡처 루프. self._stop 이 set 될 때까지 _emit() 호출."""
@staticmethod
@abc.abstractmethod
def available() -> bool:
"""현재 환경에서 이 백엔드를 쓸 수 있는가."""
@staticmethod
@abc.abstractmethod
def list_sources() -> list[AudioSource]:
"""선택 가능한 캡처 대상 목록."""
# --- 공통 동작 -----------------------------------------------------
def start(self) -> None:
if self._thread and self._thread.is_alive():
return
self._stop.clear()
self._error = None
self._thread = threading.Thread(
target=self._thread_main, name=f"capture-{self.name}", daemon=True
)
self._thread.start()
def _thread_main(self) -> None:
try:
self._run()
except Exception as exc: # noqa: BLE001 - 워커 스레드 경계
self._error = exc
def stop(self, timeout: float = 2.0) -> None:
self._stop.set()
if self._thread:
self._thread.join(timeout=timeout)
self._thread = None
@property
def running(self) -> bool:
return bool(self._thread and self._thread.is_alive())
@property
def error(self) -> Exception | None:
return self._error
def _emit(self, frame: np.ndarray) -> None:
"""캡처 프레임 투입. 큐가 가득 차면 가장 오래된 프레임을 버린다.
실시간 자막에서는 밀린 오디오보다 최신 오디오가 항상 더 가치 있다.
"""
try:
self._queue.put_nowait(frame)
except queue.Full:
try:
self._queue.get_nowait()
self._queue.put_nowait(frame)
except (queue.Empty, queue.Full):
pass
def read(self, timeout: float = 0.5) -> np.ndarray | None:
try:
return self._queue.get(timeout=timeout)
except queue.Empty:
return None
def drain(self) -> None:
while not self._queue.empty():
try:
self._queue.get_nowait()
except queue.Empty:
break
def to_mono_16k(data: np.ndarray, channels: int, src_rate: int) -> np.ndarray:
"""인터리브된 float32 PCM을 16kHz 모노로 변환."""
if data.size == 0:
return data.astype(np.float32, copy=False)
audio = data.astype(np.float32, copy=False)
if channels > 1:
usable = (audio.size // channels) * channels
audio = audio[:usable].reshape(-1, channels).mean(axis=1)
if src_rate != SAMPLE_RATE and audio.size:
# 선형 보간 리샘플. 음성인식 입력으로는 충분한 품질이며 의존성이 없다.
duration = audio.size / src_rate
target_len = max(1, int(duration * SAMPLE_RATE))
audio = np.interp(
np.linspace(0.0, audio.size - 1, target_len, dtype=np.float64),
np.arange(audio.size, dtype=np.float64),
audio,
).astype(np.float32)
return audio