"""오디오 캡처 백엔드 공통 인터페이스. 모든 백엔드는 16kHz / mono / float32(-1.0~1.0) 프레임을 큐로 흘려보낸다. 리샘플링과 다운믹스는 각 백엔드가 책임진다. """ from __future__ import annotations import abc import queue import threading from dataclasses import dataclass import numpy as np from ..constants import SAMPLE_RATE @dataclass(frozen=True) class AudioSource: """UI에 노출되는 캡처 대상 한 줄.""" kind: str # "process" | "device" identifier: str # pid 문자열 또는 장치 인덱스 label: str detail: str = "" icon_path: str = "" @property def pid(self) -> int: return int(self.identifier) if self.kind == "process" else 0 @property def device_index(self) -> int: return int(self.identifier) if self.kind == "device" else -1 class CaptureError(RuntimeError): """캡처를 시작할 수 없을 때.""" class CaptureBackend(abc.ABC): """오디오 캡처 백엔드 베이스.""" name = "base" #: 특정 프로그램 소리만 분리해서 받을 수 있는가 per_process = False def __init__(self, max_queue_frames: int = 400) -> None: self._queue: queue.Queue[np.ndarray] = queue.Queue(maxsize=max_queue_frames) self._stop = threading.Event() self._thread: threading.Thread | None = None self._error: Exception | None = None # --- 하위 클래스가 구현 ------------------------------------------- @abc.abstractmethod def _run(self) -> None: """블로킹 캡처 루프. self._stop 이 set 될 때까지 _emit() 호출.""" @staticmethod @abc.abstractmethod def available() -> bool: """현재 환경에서 이 백엔드를 쓸 수 있는가.""" @staticmethod @abc.abstractmethod def list_sources() -> list[AudioSource]: """선택 가능한 캡처 대상 목록.""" # --- 공통 동작 ----------------------------------------------------- def start(self) -> None: if self._thread and self._thread.is_alive(): return self._stop.clear() self._error = None self._thread = threading.Thread( target=self._thread_main, name=f"capture-{self.name}", daemon=True ) self._thread.start() def _thread_main(self) -> None: try: self._run() except Exception as exc: # noqa: BLE001 - 워커 스레드 경계 self._error = exc def stop(self, timeout: float = 2.0) -> None: self._stop.set() if self._thread: self._thread.join(timeout=timeout) self._thread = None @property def running(self) -> bool: return bool(self._thread and self._thread.is_alive()) @property def error(self) -> Exception | None: return self._error def _emit(self, frame: np.ndarray) -> None: """캡처 프레임 투입. 큐가 가득 차면 가장 오래된 프레임을 버린다. 실시간 자막에서는 밀린 오디오보다 최신 오디오가 항상 더 가치 있다. """ try: self._queue.put_nowait(frame) except queue.Full: try: self._queue.get_nowait() self._queue.put_nowait(frame) except (queue.Empty, queue.Full): pass def read(self, timeout: float = 0.5) -> np.ndarray | None: try: return self._queue.get(timeout=timeout) except queue.Empty: return None def drain(self) -> None: while not self._queue.empty(): try: self._queue.get_nowait() except queue.Empty: break def to_mono_16k(data: np.ndarray, channels: int, src_rate: int) -> np.ndarray: """인터리브된 float32 PCM을 16kHz 모노로 변환.""" if data.size == 0: return data.astype(np.float32, copy=False) audio = data.astype(np.float32, copy=False) if channels > 1: usable = (audio.size // channels) * channels audio = audio[:usable].reshape(-1, channels).mean(axis=1) if src_rate != SAMPLE_RATE and audio.size: # 선형 보간 리샘플. 음성인식 입력으로는 충분한 품질이며 의존성이 없다. duration = audio.size / src_rate target_len = max(1, int(duration * SAMPLE_RATE)) audio = np.interp( np.linspace(0.0, audio.size - 1, target_len, dtype=np.float64), np.arange(audio.size, dtype=np.float64), audio, ).astype(np.float32) return audio