말투 (존댓말 기본, 반말 선택) - 원문이 실제로 존댓말이면 반말 모드여도 존댓말을 지킨다. 상대가 정중하게 말했는데 자막이 반말이면 뉘앙스가 뒤집히기 때문. - 영어·중국어는 문법적 높임이 없으므로 항상 고른 모드를 따른다. "please" 를 존댓말 근거로 삼으면 오탐이 많아 쓰지 않았다. - 낮춤 변환은 한글 자모를 분해해 실제 활용 규칙(모음조화 + 축약)을 구현했다. 오+아->와, 지+어->져, 하+어->해, 았/었 뒤는 항상 어, 치겠습니다->칠게. 어미를 나열하는 방식보다 훨씬 넓게 맞는다. - 변환 방향은 존댓말->반말 한쪽만 한다. 번역 모델의 한국어 출력이 이미 격식체라 존댓말 모드는 손댈 필요가 없고, 반대 방향은 훨씬 자주 틀린다. - 지시문을 이해하는 Qwen3 에는 프롬프트로도 전달한다. Seed-X 는 지시문을 못 알아듣는 모델이라 후처리로만 맞춘다. 포터블 exe - PyInstaller 명세 + 빌드 스크립트. torch 를 의도적으로 제외했다. torch+CUDA 만 2.5GB 라 onefile 로 묶으면 실행할 때마다 그걸 임시폴더에 푸느라 1분 넘게 걸려 쓸 수 없다. - 음성인식(faster-whisper)도 번역(NLLB)도 CTranslate2 위에서 돌아 torch 가 필요 없다. 덕분에 1~3티어는 그대로 다 되고 exe 는 3GB -> 500MB 가 된다. - 4~5티어는 못 쓰므로 tier_availability() 로 판정해 모델 화면에 '사용 불가'와 이유를 표시한다. torch 없는 환경에서 앱 전체가 뜨는 것을 확인했다. AI 이미지 - SDXL-turbo 로 아이콘/배경 생성 (로컬 GPU, 피크 VRAM 1.9GB). - 글자는 AI 가 제대로 못 쓰므로 아트만 AI 로 만들고 타이포그래피는 정확히 렌더링해 합성했다. icon.ico 는 16~256px 멀티해상도. 검증: pytest 163개 통과 (말투 53개 신규), ruff clean Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
132 lines
3.7 KiB
Python
132 lines
3.7 KiB
Python
"""엔진 파이프라인을 가짜 모델로 끝까지 돌려본다 (GPU/오디오 장치 불필요)."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import time
|
|
|
|
import numpy as np
|
|
import pytest
|
|
|
|
from livesub.audio.base import CaptureBackend, to_mono_16k
|
|
from livesub.config import AppConfig
|
|
from livesub.constants import SAMPLE_RATE
|
|
from livesub.core.engine import TranslationEngine
|
|
from livesub.models.asr import Transcript
|
|
|
|
|
|
class FakeCapture(CaptureBackend):
|
|
"""말-침묵-말-침묵 패턴을 즉시 흘려보낸다."""
|
|
|
|
name = "fake"
|
|
per_process = True
|
|
|
|
@staticmethod
|
|
def available() -> bool:
|
|
return True
|
|
|
|
@staticmethod
|
|
def list_sources():
|
|
return []
|
|
|
|
def _run(self) -> None:
|
|
def tone(ms, amp=0.35):
|
|
n = SAMPLE_RATE * ms // 1000
|
|
t = np.arange(n, dtype=np.float32) / SAMPLE_RATE
|
|
return (np.sin(2 * np.pi * 240 * t) * amp).astype(np.float32)
|
|
|
|
pattern = np.concatenate(
|
|
[np.zeros(SAMPLE_RATE // 5, np.float32), tone(900),
|
|
np.zeros(SAMPLE_RATE, np.float32)]
|
|
)
|
|
while not self._stop.is_set():
|
|
for i in range(0, pattern.size, 1600):
|
|
if self._stop.is_set():
|
|
return
|
|
self._emit(pattern[i : i + 1600].copy())
|
|
time.sleep(0.005)
|
|
|
|
|
|
class FakeRecognizer:
|
|
loaded = True
|
|
|
|
def load(self):
|
|
pass
|
|
|
|
def unload(self):
|
|
pass
|
|
|
|
def transcribe(self, audio, language=None, fast=False, prompt=""):
|
|
return Transcript(text="hello world", language="en", confidence=0.99)
|
|
|
|
|
|
class FakeTranslator:
|
|
loaded = True
|
|
|
|
def load(self):
|
|
pass
|
|
|
|
def unload(self):
|
|
pass
|
|
|
|
def translate(self, text, source_lang, target_lang, glossary=None, **kwargs):
|
|
# 실제 Translator 와 같은 키워드(speech_level, source_politeness)를 받는다.
|
|
self.last_kwargs = kwargs
|
|
return f"[{target_lang}] {text}"
|
|
|
|
|
|
@pytest.fixture
|
|
def engine(tmp_path, monkeypatch):
|
|
config = AppConfig()
|
|
config.models.preload_on_start = False
|
|
config.glossary.path = str(tmp_path / "glossary.json")
|
|
config.audio.partial_interval_ms = 0
|
|
config.audio.silence_ms = 300
|
|
config.audio.min_segment_ms = 200
|
|
|
|
lines = []
|
|
eng = TranslationEngine(config, on_line=lines.append)
|
|
monkeypatch.setattr(eng.models, "recognizer", lambda: FakeRecognizer())
|
|
monkeypatch.setattr(eng.models, "translator", lambda: FakeTranslator())
|
|
monkeypatch.setattr("livesub.core.engine.create_capture", lambda cfg: FakeCapture())
|
|
eng.lines = lines
|
|
yield eng
|
|
eng.shutdown()
|
|
|
|
|
|
def test_engine_produces_translated_lines(engine):
|
|
engine.start()
|
|
deadline = time.time() + 12
|
|
while time.time() < deadline and not engine.lines:
|
|
time.sleep(0.05)
|
|
engine.stop()
|
|
|
|
assert engine.lines, "12초 안에 자막이 한 줄도 나오지 않았습니다"
|
|
line = engine.lines[0]
|
|
assert line.source_text == "hello world"
|
|
assert line.translated_text == "[ko] hello world"
|
|
assert line.source_lang == "en"
|
|
assert line.is_final
|
|
assert line.latency_s >= 0
|
|
|
|
|
|
def test_stop_is_idempotent(engine):
|
|
engine.start()
|
|
time.sleep(0.3)
|
|
engine.stop()
|
|
engine.stop()
|
|
assert not engine.running
|
|
|
|
|
|
def test_to_mono_16k_downmixes_and_resamples():
|
|
stereo_48k = np.tile(np.array([1.0, -1.0], dtype=np.float32), 48_000)
|
|
out = to_mono_16k(stereo_48k, channels=2, src_rate=48_000)
|
|
assert out.dtype == np.float32
|
|
assert abs(out.size - 16_000) <= 2 # 1초 분량
|
|
assert np.allclose(out, 0.0, atol=1e-6) # L+R 이 상쇄
|
|
|
|
|
|
def test_to_mono_16k_passthrough_when_already_correct():
|
|
mono = np.linspace(-1, 1, 16_000, dtype=np.float32)
|
|
out = to_mono_16k(mono, channels=1, src_rate=16_000)
|
|
assert np.array_equal(out, mono)
|