feat: Hearo 1차 구현 — 프로그램 소리 실시간 번역 자막
프로그램별 오디오를 캡처해 로컬 GPU에서 음성인식→번역하고 화면 위 자막으로 보여주는 데스크톱 앱. 한/영/일/중 4개 언어. 구성 - audio: WASAPI 프로그램별 캡처(C++ 보조 프로그램) + 장치 루프백 폴백, 적응형 VAD 발화 분할 - models: 속도~품질 5단계 티어, faster-whisper + CTranslate2/LLM 2백엔드, 용어집(플레이스홀더 보호 + 프롬프트 주입) - core: Qt 비의존 파이프라인 엔진 (캡처/분할/추론 3스레드, 큐 연결) - ui: 사이드바 5화면 + 무테두리 항상위 자막 오버레이, 자체 다크 테마 모델 선정 근거는 docs/MODELS.md, 추가학습 가능 여부와 방법은 docs/FINETUNING.md 참고. 검증: pytest 39개 통과 (GPU·오디오 장치 없이 실행), ruff clean Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
129
tests/test_pipeline.py
Normal file
129
tests/test_pipeline.py
Normal file
@@ -0,0 +1,129 @@
|
||||
"""엔진 파이프라인을 가짜 모델로 끝까지 돌려본다 (GPU/오디오 장치 불필요)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
from hearo.audio.base import CaptureBackend, to_mono_16k
|
||||
from hearo.config import AppConfig
|
||||
from hearo.constants import SAMPLE_RATE
|
||||
from hearo.core.engine import TranslationEngine
|
||||
from hearo.models.asr import Transcript
|
||||
|
||||
|
||||
class FakeCapture(CaptureBackend):
|
||||
"""말-침묵-말-침묵 패턴을 즉시 흘려보낸다."""
|
||||
|
||||
name = "fake"
|
||||
per_process = True
|
||||
|
||||
@staticmethod
|
||||
def available() -> bool:
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def list_sources():
|
||||
return []
|
||||
|
||||
def _run(self) -> None:
|
||||
def tone(ms, amp=0.35):
|
||||
n = SAMPLE_RATE * ms // 1000
|
||||
t = np.arange(n, dtype=np.float32) / SAMPLE_RATE
|
||||
return (np.sin(2 * np.pi * 240 * t) * amp).astype(np.float32)
|
||||
|
||||
pattern = np.concatenate(
|
||||
[np.zeros(SAMPLE_RATE // 5, np.float32), tone(900),
|
||||
np.zeros(SAMPLE_RATE, np.float32)]
|
||||
)
|
||||
while not self._stop.is_set():
|
||||
for i in range(0, pattern.size, 1600):
|
||||
if self._stop.is_set():
|
||||
return
|
||||
self._emit(pattern[i : i + 1600].copy())
|
||||
time.sleep(0.005)
|
||||
|
||||
|
||||
class FakeRecognizer:
|
||||
loaded = True
|
||||
|
||||
def load(self):
|
||||
pass
|
||||
|
||||
def unload(self):
|
||||
pass
|
||||
|
||||
def transcribe(self, audio, language=None, fast=False, prompt=""):
|
||||
return Transcript(text="hello world", language="en", confidence=0.99)
|
||||
|
||||
|
||||
class FakeTranslator:
|
||||
loaded = True
|
||||
|
||||
def load(self):
|
||||
pass
|
||||
|
||||
def unload(self):
|
||||
pass
|
||||
|
||||
def translate(self, text, source_lang, target_lang, glossary=None):
|
||||
return f"[{target_lang}] {text}"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def engine(tmp_path, monkeypatch):
|
||||
config = AppConfig()
|
||||
config.models.preload_on_start = False
|
||||
config.glossary.path = str(tmp_path / "glossary.json")
|
||||
config.audio.partial_interval_ms = 0
|
||||
config.audio.silence_ms = 300
|
||||
config.audio.min_segment_ms = 200
|
||||
|
||||
lines = []
|
||||
eng = TranslationEngine(config, on_line=lines.append)
|
||||
monkeypatch.setattr(eng.models, "recognizer", lambda: FakeRecognizer())
|
||||
monkeypatch.setattr(eng.models, "translator", lambda: FakeTranslator())
|
||||
monkeypatch.setattr("hearo.core.engine.create_capture", lambda cfg: FakeCapture())
|
||||
eng.lines = lines
|
||||
yield eng
|
||||
eng.shutdown()
|
||||
|
||||
|
||||
def test_engine_produces_translated_lines(engine):
|
||||
engine.start()
|
||||
deadline = time.time() + 12
|
||||
while time.time() < deadline and not engine.lines:
|
||||
time.sleep(0.05)
|
||||
engine.stop()
|
||||
|
||||
assert engine.lines, "12초 안에 자막이 한 줄도 나오지 않았습니다"
|
||||
line = engine.lines[0]
|
||||
assert line.source_text == "hello world"
|
||||
assert line.translated_text == "[ko] hello world"
|
||||
assert line.source_lang == "en"
|
||||
assert line.is_final
|
||||
assert line.latency_s >= 0
|
||||
|
||||
|
||||
def test_stop_is_idempotent(engine):
|
||||
engine.start()
|
||||
time.sleep(0.3)
|
||||
engine.stop()
|
||||
engine.stop()
|
||||
assert not engine.running
|
||||
|
||||
|
||||
def test_to_mono_16k_downmixes_and_resamples():
|
||||
stereo_48k = np.tile(np.array([1.0, -1.0], dtype=np.float32), 48_000)
|
||||
out = to_mono_16k(stereo_48k, channels=2, src_rate=48_000)
|
||||
assert out.dtype == np.float32
|
||||
assert abs(out.size - 16_000) <= 2 # 1초 분량
|
||||
assert np.allclose(out, 0.0, atol=1e-6) # L+R 이 상쇄
|
||||
|
||||
|
||||
def test_to_mono_16k_passthrough_when_already_correct():
|
||||
mono = np.linspace(-1, 1, 16_000, dtype=np.float32)
|
||||
out = to_mono_16k(mono, channels=1, src_rate=16_000)
|
||||
assert np.array_equal(out, mono)
|
||||
Reference in New Issue
Block a user