Files
live-app-translator/tests/test_prompts.py
EJClaw 8b36b244ff feat: LiveSub 2차 — 자막 on/off, 모니터·위치 선택, GPU 저부하, 게임 용어집
이름을 Hearo → LiveSub 로 변경. 말장난보다 하는 일이 바로 보이는 쪽이 낫다.

자막 on/off
- 전역 단축키 Ctrl+Alt+S (자막) / Ctrl+Alt+D (번역) — Windows RegisterHotKey +
  네이티브 이벤트 필터라 게임 창이 떠 있어도 동작. 다른 OS 에서는 no-op
- 단축키·체크박스·우클릭 메뉴가 모두 같은 경로를 타도록 통합

모니터 선택 + 디스코드식 배치
- placement.py: Qt 비의존 배치 계산 (모니터 목록 → 9분할 좌표)
- AnchorGrid 위젯으로 3x3 위치 선택, 드래그하면 자유 배치로 전환
- 모니터 구성이 바뀌어도 자막이 화면 밖으로 사라지지 않도록 클램프

GPU 저부하 모드 (기본 켜짐)
- 연산 정밀도 int8 강등, VRAM 상한 35%, 추론 후 60ms 양보,
  300초 무음 시 모델 언로드, 중간 결과 비활성
- 기본 티어를 균형(6GB) → 신속(4GB) 으로 하향
- 8GB GPU 기준 자막 2.8GB / 게임 5.2GB

게임 용어집 번들 (304개)
- 롤 74 · 발로란트 59 · 오버워치2 49 · FPS공통 46 · 마크 38 · 방송 38
- 체크박스로 켜고 끄며, 사용자가 직접 등록한 항목이 항상 우선

검증: pytest 91개 통과 (배치 14 + 팩 12 + UI 6 신규), ruff clean

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-09-21 17:49:26 +09:00

83 lines
3.3 KiB
Python

"""번역 프롬프트 형식 회귀 테스트.
Seed-X 는 chat template 없는 번역 전용 completion 모델이라 모델 카드가 정한
형식을 한 글자도 벗어나면 안 된다. 특히 끝의 `<언어코드>` 태그는 PPO 학습에
쓰인 것이라 빠지면 품질이 무너진다.
"""
from __future__ import annotations
import pytest
from livesub.constants import LANGUAGE_CODES, LANGUAGES
from livesub.models.glossary import Glossary, GlossaryEntry
from livesub.models.tiers import TIERS, MTBackend, PromptStyle, get_tier
from livesub.models.translator import build_seedx_prompt
def test_seedx_prompt_matches_model_card_exactly():
"""모델 카드 예시: "Translate the following English sentence into Chinese:\\nMay the force be with you <zh>" """
assert build_seedx_prompt("May the force be with you", "en", "zh") == (
"Translate the following English sentence into Chinese:\n"
"May the force be with you <zh>"
)
def test_seedx_prompt_always_ends_with_target_language_tag():
for target in LANGUAGE_CODES:
prompt = build_seedx_prompt("hello", "en", target)
assert prompt.endswith(f" <{target}>"), f"{target} 태그 누락"
def test_seedx_prompt_has_no_extra_instructions():
"""지시문을 끼워 넣으면 Seed-X 의 학습 분포를 벗어난다."""
prompt = build_seedx_prompt("fall back now", "en", "ko")
lowered = prompt.lower()
for forbidden in ("casual", "output only", "glossary", "game or broadcast", "용어"):
assert forbidden not in lowered
assert prompt.count("\n") == 1 # 지시문 한 줄 + 본문 한 줄
@pytest.mark.parametrize("code", LANGUAGE_CODES)
def test_every_language_has_a_seedx_tag(code):
assert LANGUAGES[code]["seedx"], f"{code} 에 seedx 태그가 없습니다"
def test_seedx_tier_does_not_use_prompt_glossary():
"""Seed-X 는 지시문을 못 알아들으므로 프롬프트 용어집을 쓰면 안 된다."""
precision = get_tier("precision")
assert precision.mt.prompt_style is PromptStyle.SEEDX
assert precision.mt.supports_prompt_glossary is False
def test_instruct_tier_uses_prompt_glossary():
ultimate = get_tier("ultimate")
assert ultimate.mt.prompt_style is PromptStyle.INSTRUCT
assert ultimate.mt.supports_prompt_glossary is True
def test_seq2seq_tiers_never_use_prompt_glossary():
for tier in TIERS.values():
if tier.mt.backend is MTBackend.CTRANSLATE2:
assert tier.mt.prompt_style is PromptStyle.NONE
assert tier.mt.supports_prompt_glossary is False
def test_every_llm_tier_declares_a_prompt_style():
for tier in TIERS.values():
if tier.mt.backend is MTBackend.TRANSFORMERS:
assert tier.mt.prompt_style is not PromptStyle.NONE, tier.key
def test_glossary_survives_seedx_placeholder_path():
"""Seed-X 경로에서 용어집이 프롬프트가 아니라 치환으로 동작하는지."""
g = Glossary([GlossaryEntry("Nexus", {"ko": "넥서스"})])
protected, repl = g.protect("push to the Nexus", "ko")
prompt = build_seedx_prompt(protected, "en", "ko")
assert "Nexus" not in prompt # 모델이 건드릴 수 없게 가려짐
assert prompt.endswith(" <ko>")
# 모델이 플레이스홀더를 그대로 통과시켰다고 가정
assert Glossary.restore("⟦0⟧로 밀어", repl) == "넥서스로 밀어"