feat: Hearo 1차 구현 — 프로그램 소리 실시간 번역 자막
프로그램별 오디오를 캡처해 로컬 GPU에서 음성인식→번역하고 화면 위 자막으로 보여주는 데스크톱 앱. 한/영/일/중 4개 언어. 구성 - audio: WASAPI 프로그램별 캡처(C++ 보조 프로그램) + 장치 루프백 폴백, 적응형 VAD 발화 분할 - models: 속도~품질 5단계 티어, faster-whisper + CTranslate2/LLM 2백엔드, 용어집(플레이스홀더 보호 + 프롬프트 주입) - core: Qt 비의존 파이프라인 엔진 (캡처/분할/추론 3스레드, 큐 연결) - ui: 사이드바 5화면 + 무테두리 항상위 자막 오버레이, 자체 다크 테마 모델 선정 근거는 docs/MODELS.md, 추가학습 가능 여부와 방법은 docs/FINETUNING.md 참고. 검증: pytest 39개 통과 (GPU·오디오 장치 없이 실행), ruff clean Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
58
tests/test_config.py
Normal file
58
tests/test_config.py
Normal file
@@ -0,0 +1,58 @@
|
||||
import json
|
||||
|
||||
from hearo.config import AppConfig
|
||||
from hearo.models.tiers import DEFAULT_TIER
|
||||
|
||||
|
||||
def test_defaults_are_sane():
|
||||
cfg = AppConfig()
|
||||
assert cfg.models.tier == DEFAULT_TIER
|
||||
assert cfg.models.target_lang == "ko"
|
||||
assert cfg.subtitle.font_size > 0
|
||||
assert cfg.audio.silence_ms > 0
|
||||
|
||||
|
||||
def test_save_load_roundtrip(tmp_path):
|
||||
path = tmp_path / "config.json"
|
||||
cfg = AppConfig()
|
||||
cfg.subtitle.font_size = 52
|
||||
cfg.subtitle.text_color = "#FF00AA"
|
||||
cfg.models.tier = "precision"
|
||||
cfg.audio.target_pid = 4242
|
||||
cfg.save(path)
|
||||
|
||||
loaded = AppConfig.load(path)
|
||||
assert loaded.subtitle.font_size == 52
|
||||
assert loaded.subtitle.text_color == "#FF00AA"
|
||||
assert loaded.models.tier == "precision"
|
||||
assert loaded.audio.target_pid == 4242
|
||||
|
||||
|
||||
def test_missing_file_gives_defaults(tmp_path):
|
||||
assert AppConfig.load(tmp_path / "nope.json").models.tier == DEFAULT_TIER
|
||||
|
||||
|
||||
def test_corrupt_file_gives_defaults(tmp_path):
|
||||
path = tmp_path / "config.json"
|
||||
path.write_text("{ this is not json", encoding="utf-8")
|
||||
assert AppConfig.load(path).models.tier == DEFAULT_TIER
|
||||
|
||||
|
||||
def test_unknown_keys_are_ignored_and_missing_keys_defaulted(tmp_path):
|
||||
"""스키마가 바뀌어도 사용자의 기존 설정이 깨지면 안 된다."""
|
||||
path = tmp_path / "config.json"
|
||||
path.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"theme": "light",
|
||||
"removed_option": True,
|
||||
"subtitle": {"font_size": 40, "gone_field": 1},
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
cfg = AppConfig.load(path)
|
||||
assert cfg.theme == "light"
|
||||
assert cfg.subtitle.font_size == 40
|
||||
assert cfg.subtitle.bold is True # 기본값 유지
|
||||
assert not hasattr(cfg, "removed_option")
|
||||
75
tests/test_glossary.py
Normal file
75
tests/test_glossary.py
Normal file
@@ -0,0 +1,75 @@
|
||||
from hearo.models.glossary import Glossary, GlossaryEntry
|
||||
|
||||
|
||||
def make() -> Glossary:
|
||||
return Glossary(
|
||||
[
|
||||
GlossaryEntry("headshot", {"ko": "헤드샷"}),
|
||||
GlossaryEntry("head", {"ko": "머리"}),
|
||||
GlossaryEntry("Nexus", {"ko": "넥서스", "ja": "ネクサス"}),
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
def test_protect_and_restore_roundtrip():
|
||||
g = make()
|
||||
protected, repl = g.protect("Nice headshot on the Nexus", "ko")
|
||||
assert "headshot" not in protected
|
||||
assert "Nexus" not in protected
|
||||
assert repl == ["헤드샷", "넥서스"]
|
||||
# 번역 모델이 플레이스홀더를 그대로 통과시켰다고 가정
|
||||
assert Glossary.restore(protected, repl) == "Nice 헤드샷 on the 넥서스"
|
||||
|
||||
|
||||
def test_longest_match_wins():
|
||||
"""'headshot'이 'head'에 먼저 잡아먹히면 안 된다."""
|
||||
g = make()
|
||||
_, repl = g.protect("headshot", "ko")
|
||||
assert repl == ["헤드샷"]
|
||||
|
||||
|
||||
def test_word_boundary_for_latin_terms():
|
||||
g = make()
|
||||
_, repl = g.protect("overheated", "ko")
|
||||
assert repl == []
|
||||
|
||||
|
||||
def test_missing_target_language_is_left_alone():
|
||||
g = make()
|
||||
protected, repl = g.protect("headshot", "ja")
|
||||
assert protected == "headshot"
|
||||
assert repl == []
|
||||
|
||||
|
||||
def test_prompt_hint_only_includes_present_terms():
|
||||
g = make()
|
||||
hint = g.prompt_hint("push to the Nexus", "ko")
|
||||
assert "Nexus -> 넥서스" in hint
|
||||
assert "headshot" not in hint
|
||||
assert g.prompt_hint("nothing here", "ko") == ""
|
||||
|
||||
|
||||
def test_case_insensitive_by_default():
|
||||
g = make()
|
||||
_, repl = g.protect("NEXUS down", "ko")
|
||||
assert repl == ["넥서스"]
|
||||
|
||||
|
||||
def test_save_and_load(tmp_path):
|
||||
path = tmp_path / "glossary.json"
|
||||
make().save(path)
|
||||
loaded = Glossary.load(path)
|
||||
assert len(loaded) == 3
|
||||
assert loaded.entries[2].target_for("ja") == "ネクサス"
|
||||
|
||||
|
||||
def test_add_replaces_existing_entry():
|
||||
g = make()
|
||||
g.add(GlossaryEntry("Nexus", {"ko": "본진"}))
|
||||
assert len(g) == 3
|
||||
_, repl = g.protect("Nexus", "ko")
|
||||
assert repl == ["본진"]
|
||||
|
||||
|
||||
def test_restore_ignores_out_of_range_placeholder():
|
||||
assert Glossary.restore("a ⟦5⟧ b", ["x"]) == "a b"
|
||||
129
tests/test_pipeline.py
Normal file
129
tests/test_pipeline.py
Normal file
@@ -0,0 +1,129 @@
|
||||
"""엔진 파이프라인을 가짜 모델로 끝까지 돌려본다 (GPU/오디오 장치 불필요)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
from hearo.audio.base import CaptureBackend, to_mono_16k
|
||||
from hearo.config import AppConfig
|
||||
from hearo.constants import SAMPLE_RATE
|
||||
from hearo.core.engine import TranslationEngine
|
||||
from hearo.models.asr import Transcript
|
||||
|
||||
|
||||
class FakeCapture(CaptureBackend):
|
||||
"""말-침묵-말-침묵 패턴을 즉시 흘려보낸다."""
|
||||
|
||||
name = "fake"
|
||||
per_process = True
|
||||
|
||||
@staticmethod
|
||||
def available() -> bool:
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def list_sources():
|
||||
return []
|
||||
|
||||
def _run(self) -> None:
|
||||
def tone(ms, amp=0.35):
|
||||
n = SAMPLE_RATE * ms // 1000
|
||||
t = np.arange(n, dtype=np.float32) / SAMPLE_RATE
|
||||
return (np.sin(2 * np.pi * 240 * t) * amp).astype(np.float32)
|
||||
|
||||
pattern = np.concatenate(
|
||||
[np.zeros(SAMPLE_RATE // 5, np.float32), tone(900),
|
||||
np.zeros(SAMPLE_RATE, np.float32)]
|
||||
)
|
||||
while not self._stop.is_set():
|
||||
for i in range(0, pattern.size, 1600):
|
||||
if self._stop.is_set():
|
||||
return
|
||||
self._emit(pattern[i : i + 1600].copy())
|
||||
time.sleep(0.005)
|
||||
|
||||
|
||||
class FakeRecognizer:
|
||||
loaded = True
|
||||
|
||||
def load(self):
|
||||
pass
|
||||
|
||||
def unload(self):
|
||||
pass
|
||||
|
||||
def transcribe(self, audio, language=None, fast=False, prompt=""):
|
||||
return Transcript(text="hello world", language="en", confidence=0.99)
|
||||
|
||||
|
||||
class FakeTranslator:
|
||||
loaded = True
|
||||
|
||||
def load(self):
|
||||
pass
|
||||
|
||||
def unload(self):
|
||||
pass
|
||||
|
||||
def translate(self, text, source_lang, target_lang, glossary=None):
|
||||
return f"[{target_lang}] {text}"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def engine(tmp_path, monkeypatch):
|
||||
config = AppConfig()
|
||||
config.models.preload_on_start = False
|
||||
config.glossary.path = str(tmp_path / "glossary.json")
|
||||
config.audio.partial_interval_ms = 0
|
||||
config.audio.silence_ms = 300
|
||||
config.audio.min_segment_ms = 200
|
||||
|
||||
lines = []
|
||||
eng = TranslationEngine(config, on_line=lines.append)
|
||||
monkeypatch.setattr(eng.models, "recognizer", lambda: FakeRecognizer())
|
||||
monkeypatch.setattr(eng.models, "translator", lambda: FakeTranslator())
|
||||
monkeypatch.setattr("hearo.core.engine.create_capture", lambda cfg: FakeCapture())
|
||||
eng.lines = lines
|
||||
yield eng
|
||||
eng.shutdown()
|
||||
|
||||
|
||||
def test_engine_produces_translated_lines(engine):
|
||||
engine.start()
|
||||
deadline = time.time() + 12
|
||||
while time.time() < deadline and not engine.lines:
|
||||
time.sleep(0.05)
|
||||
engine.stop()
|
||||
|
||||
assert engine.lines, "12초 안에 자막이 한 줄도 나오지 않았습니다"
|
||||
line = engine.lines[0]
|
||||
assert line.source_text == "hello world"
|
||||
assert line.translated_text == "[ko] hello world"
|
||||
assert line.source_lang == "en"
|
||||
assert line.is_final
|
||||
assert line.latency_s >= 0
|
||||
|
||||
|
||||
def test_stop_is_idempotent(engine):
|
||||
engine.start()
|
||||
time.sleep(0.3)
|
||||
engine.stop()
|
||||
engine.stop()
|
||||
assert not engine.running
|
||||
|
||||
|
||||
def test_to_mono_16k_downmixes_and_resamples():
|
||||
stereo_48k = np.tile(np.array([1.0, -1.0], dtype=np.float32), 48_000)
|
||||
out = to_mono_16k(stereo_48k, channels=2, src_rate=48_000)
|
||||
assert out.dtype == np.float32
|
||||
assert abs(out.size - 16_000) <= 2 # 1초 분량
|
||||
assert np.allclose(out, 0.0, atol=1e-6) # L+R 이 상쇄
|
||||
|
||||
|
||||
def test_to_mono_16k_passthrough_when_already_correct():
|
||||
mono = np.linspace(-1, 1, 16_000, dtype=np.float32)
|
||||
out = to_mono_16k(mono, channels=1, src_rate=16_000)
|
||||
assert np.array_equal(out, mono)
|
||||
78
tests/test_segmenter.py
Normal file
78
tests/test_segmenter.py
Normal file
@@ -0,0 +1,78 @@
|
||||
import numpy as np
|
||||
|
||||
from hearo.audio.segmenter import Segmenter, SegmenterConfig
|
||||
from hearo.constants import SAMPLE_RATE
|
||||
|
||||
|
||||
def tone(ms: int, amplitude: float = 0.3) -> np.ndarray:
|
||||
n = SAMPLE_RATE * ms // 1000
|
||||
t = np.arange(n, dtype=np.float32) / SAMPLE_RATE
|
||||
return (np.sin(2 * np.pi * 220 * t) * amplitude).astype(np.float32)
|
||||
|
||||
|
||||
def silence(ms: int) -> np.ndarray:
|
||||
return np.zeros(SAMPLE_RATE * ms // 1000, dtype=np.float32)
|
||||
|
||||
|
||||
def config(**kw) -> SegmenterConfig:
|
||||
base = dict(silence_ms=300, min_segment_ms=200, max_segment_ms=5000,
|
||||
partial_interval_ms=0)
|
||||
base.update(kw)
|
||||
return SegmenterConfig(**base)
|
||||
|
||||
|
||||
def test_speech_then_silence_emits_one_final_segment():
|
||||
seg = Segmenter(config())
|
||||
out = seg.push(silence(200))
|
||||
assert out == []
|
||||
out += seg.push(tone(800))
|
||||
out += seg.push(silence(600))
|
||||
finals = [s for s in out if s.is_final]
|
||||
assert len(finals) == 1
|
||||
assert 0.7 <= finals[0].duration_s <= 1.8
|
||||
|
||||
|
||||
def test_short_blip_is_discarded():
|
||||
seg = Segmenter(config(min_segment_ms=500))
|
||||
out = seg.push(tone(60)) + seg.push(silence(600))
|
||||
assert out == []
|
||||
|
||||
|
||||
def test_max_length_forces_a_cut():
|
||||
seg = Segmenter(config(max_segment_ms=1000))
|
||||
out = seg.push(tone(3000))
|
||||
assert len([s for s in out if s.is_final]) >= 2
|
||||
|
||||
|
||||
def test_partial_results_are_emitted_while_speaking():
|
||||
seg = Segmenter(config(partial_interval_ms=300))
|
||||
out = seg.push(tone(1500))
|
||||
partials = [s for s in out if not s.is_final]
|
||||
assert len(partials) >= 2
|
||||
# 중간 결과는 누적된다
|
||||
assert partials[-1].duration_s > partials[0].duration_s
|
||||
|
||||
|
||||
def test_flush_returns_pending_speech():
|
||||
seg = Segmenter(config())
|
||||
assert seg.push(tone(700)) == [] # 아직 침묵이 안 왔으니 확정 없음
|
||||
leftover = seg.flush()
|
||||
assert leftover is not None and leftover.is_final
|
||||
assert seg.flush() is None
|
||||
|
||||
|
||||
def test_continuous_silence_never_emits():
|
||||
seg = Segmenter(config())
|
||||
for _ in range(10):
|
||||
assert seg.push(silence(200)) == []
|
||||
|
||||
|
||||
def test_lead_in_is_prepended():
|
||||
"""발화 직전 오디오가 붙어 첫 음절이 잘리지 않아야 한다."""
|
||||
seg = Segmenter(config(lead_in_ms=200))
|
||||
seg.push(silence(400))
|
||||
out = seg.push(tone(600)) + seg.push(silence(600))
|
||||
finals = [s for s in out if s.is_final]
|
||||
assert len(finals) == 1
|
||||
# 0.6초 발화 + 최대 0.2초 리드인
|
||||
assert finals[0].duration_s > 0.6
|
||||
36
tests/test_tiers.py
Normal file
36
tests/test_tiers.py
Normal file
@@ -0,0 +1,36 @@
|
||||
from hearo.models.tiers import DEFAULT_TIER, TIERS, MTBackend, get_tier, ordered_tiers
|
||||
|
||||
|
||||
def test_exactly_five_tiers_ordered_by_quality():
|
||||
tiers = ordered_tiers()
|
||||
assert len(tiers) == 5
|
||||
assert [t.order for t in tiers] == [1, 2, 3, 4, 5]
|
||||
|
||||
|
||||
def test_latency_and_vram_increase_monotonically():
|
||||
tiers = ordered_tiers()
|
||||
latencies = [t.approx_latency_s for t in tiers]
|
||||
vram = [t.min_vram_gb for t in tiers]
|
||||
assert latencies == sorted(latencies)
|
||||
assert vram == sorted(vram)
|
||||
|
||||
|
||||
def test_exactly_one_recommended_and_one_finetune_pick():
|
||||
assert sum(t.recommended for t in TIERS.values()) == 1
|
||||
assert sum(t.best_after_finetune for t in TIERS.values()) == 1
|
||||
|
||||
|
||||
def test_llm_backends_support_prompt_glossary():
|
||||
for tier in TIERS.values():
|
||||
if tier.mt.backend is MTBackend.TRANSFORMERS:
|
||||
assert tier.mt.supports_prompt_glossary
|
||||
else:
|
||||
assert not tier.mt.supports_prompt_glossary
|
||||
|
||||
|
||||
def test_unknown_key_falls_back_to_default():
|
||||
assert get_tier("does-not-exist").key == DEFAULT_TIER
|
||||
|
||||
|
||||
def test_default_tier_exists():
|
||||
assert DEFAULT_TIER in TIERS
|
||||
102
tests/test_ui_smoke.py
Normal file
102
tests/test_ui_smoke.py
Normal file
@@ -0,0 +1,102 @@
|
||||
"""UI가 실제로 만들어지고 자막이 그려지는지 확인한다 (offscreen 렌더링)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
pytest.importorskip("PySide6")
|
||||
os.environ.setdefault("QT_QPA_PLATFORM", "offscreen")
|
||||
|
||||
from PySide6.QtGui import QImage, QPainter # noqa: E402
|
||||
from PySide6.QtWidgets import QApplication # noqa: E402
|
||||
|
||||
from hearo.config import AppConfig # noqa: E402
|
||||
from hearo.core.events import EngineState, EngineStatus, TranslationLine # noqa: E402
|
||||
from hearo.ui.main_window import MainWindow # noqa: E402
|
||||
from hearo.ui.overlay import paint_subtitle # noqa: E402
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def qapp():
|
||||
app = QApplication.instance() or QApplication([])
|
||||
yield app
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def window(qapp, tmp_path, monkeypatch):
|
||||
monkeypatch.setattr("hearo.constants.user_data_dir", lambda: tmp_path)
|
||||
config = AppConfig()
|
||||
config.models.preload_on_start = False
|
||||
config.glossary.path = str(tmp_path / "glossary.json")
|
||||
win = MainWindow(config)
|
||||
yield win
|
||||
win.engine.shutdown()
|
||||
win.overlay.close()
|
||||
win.close()
|
||||
|
||||
|
||||
def test_main_window_builds_all_pages(window):
|
||||
assert window.stack.count() == 5
|
||||
|
||||
|
||||
def test_navigation_switches_pages(window):
|
||||
for index in range(5):
|
||||
window._goto_page(index)
|
||||
assert window.stack.currentIndex() == index
|
||||
|
||||
|
||||
def test_status_updates_toggle_button(window):
|
||||
window._on_status(EngineStatus(EngineState.RUNNING, "듣는 중"))
|
||||
assert window.home_page.toggle_button.text() == "번역 정지"
|
||||
window._on_status(EngineStatus(EngineState.IDLE, "대기 중"))
|
||||
assert window.home_page.toggle_button.text() == "번역 시작"
|
||||
|
||||
|
||||
def test_final_line_reaches_overlay_and_log(window):
|
||||
window.config.log_transcripts = False
|
||||
window._on_line(
|
||||
TranslationLine("hello", "안녕하세요", "en", "ko", is_final=True, latency_s=1.2)
|
||||
)
|
||||
assert window.overlay._final_text == "안녕하세요"
|
||||
assert window.home_page.log_list.count() == 1
|
||||
|
||||
|
||||
def test_partial_line_is_not_logged(window):
|
||||
window.config.log_transcripts = False
|
||||
window._on_line(TranslationLine("hel", "", "en", "ko", is_final=False))
|
||||
assert window.home_page.log_list.count() == 0
|
||||
|
||||
|
||||
def test_tier_selection_updates_config(window):
|
||||
window.models_page._on_tier_selected(window.models_page._cards[3].tier)
|
||||
assert window.config.models.tier == "precision"
|
||||
|
||||
|
||||
def test_subtitle_style_edit_propagates(window):
|
||||
window.subtitle_page.size_spin.setValue(60)
|
||||
assert window.config.subtitle.font_size == 60
|
||||
|
||||
|
||||
def test_paint_subtitle_actually_draws_pixels(qapp):
|
||||
"""자막 글씨가 투명 이미지 위에 실제로 그려지는지 (빈 화면이면 실패).
|
||||
|
||||
배경 상자가 픽셀을 채워버리면 글씨 유무를 알 수 없으므로 배경은 끄고 잰다.
|
||||
"""
|
||||
st = AppConfig().subtitle
|
||||
st.background_opacity = 0
|
||||
image = QImage(900, 200, QImage.Format.Format_ARGB32)
|
||||
image.fill(0)
|
||||
|
||||
painter = QPainter(image)
|
||||
paint_subtitle(painter, image.rect(), st, "안녕하세요 반갑습니다", "Hello there")
|
||||
painter.end()
|
||||
|
||||
non_transparent = sum(
|
||||
1
|
||||
for y in range(0, image.height(), 4)
|
||||
for x in range(0, image.width(), 4)
|
||||
if image.pixelColor(x, y).alpha() > 0
|
||||
)
|
||||
assert non_transparent > 100
|
||||
Reference in New Issue
Block a user