feat: 존댓말/반말 모드, 포터블 exe 빌드, AI 로고/배너
말투 (존댓말 기본, 반말 선택) - 원문이 실제로 존댓말이면 반말 모드여도 존댓말을 지킨다. 상대가 정중하게 말했는데 자막이 반말이면 뉘앙스가 뒤집히기 때문. - 영어·중국어는 문법적 높임이 없으므로 항상 고른 모드를 따른다. "please" 를 존댓말 근거로 삼으면 오탐이 많아 쓰지 않았다. - 낮춤 변환은 한글 자모를 분해해 실제 활용 규칙(모음조화 + 축약)을 구현했다. 오+아->와, 지+어->져, 하+어->해, 았/었 뒤는 항상 어, 치겠습니다->칠게. 어미를 나열하는 방식보다 훨씬 넓게 맞는다. - 변환 방향은 존댓말->반말 한쪽만 한다. 번역 모델의 한국어 출력이 이미 격식체라 존댓말 모드는 손댈 필요가 없고, 반대 방향은 훨씬 자주 틀린다. - 지시문을 이해하는 Qwen3 에는 프롬프트로도 전달한다. Seed-X 는 지시문을 못 알아듣는 모델이라 후처리로만 맞춘다. 포터블 exe - PyInstaller 명세 + 빌드 스크립트. torch 를 의도적으로 제외했다. torch+CUDA 만 2.5GB 라 onefile 로 묶으면 실행할 때마다 그걸 임시폴더에 푸느라 1분 넘게 걸려 쓸 수 없다. - 음성인식(faster-whisper)도 번역(NLLB)도 CTranslate2 위에서 돌아 torch 가 필요 없다. 덕분에 1~3티어는 그대로 다 되고 exe 는 3GB -> 500MB 가 된다. - 4~5티어는 못 쓰므로 tier_availability() 로 판정해 모델 화면에 '사용 불가'와 이유를 표시한다. torch 없는 환경에서 앱 전체가 뜨는 것을 확인했다. AI 이미지 - SDXL-turbo 로 아이콘/배경 생성 (로컬 GPU, 피크 VRAM 1.9GB). - 글자는 AI 가 제대로 못 쓰므로 아트만 AI 로 만들고 타이포그래피는 정확히 렌더링해 합성했다. icon.ico 는 16~256px 멀티해상도. 검증: pytest 163개 통과 (말투 53개 신규), ruff clean Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
@@ -68,7 +68,9 @@ class FakeTranslator:
|
||||
def unload(self):
|
||||
pass
|
||||
|
||||
def translate(self, text, source_lang, target_lang, glossary=None):
|
||||
def translate(self, text, source_lang, target_lang, glossary=None, **kwargs):
|
||||
# 실제 Translator 와 같은 키워드(speech_level, source_politeness)를 받는다.
|
||||
self.last_kwargs = kwargs
|
||||
return f"[{target_lang}] {text}"
|
||||
|
||||
|
||||
|
||||
242
tests/test_speech_level.py
Normal file
242
tests/test_speech_level.py
Normal file
@@ -0,0 +1,242 @@
|
||||
"""말투(존댓말/반말) 모드.
|
||||
|
||||
사용자 규칙:
|
||||
- 기본은 존댓말.
|
||||
- 높임이 없거나(영어·중국어) 알 수 없으면 고른 모드를 따른다.
|
||||
- **원문이 실제로 존댓말이면 반말 모드여도 존댓말을 지킨다.**
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from livesub.models.speech_level import (
|
||||
Politeness,
|
||||
SpeechLevel,
|
||||
apply_speech_level,
|
||||
detect_politeness,
|
||||
prompt_instruction,
|
||||
to_casual,
|
||||
)
|
||||
|
||||
# --- 원문 높임 감지 ---------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"text",
|
||||
["안녕하세요", "같이 가시죠", "도와주세요", "감사합니다", "지금 가겠습니다", "괜찮으세요?"],
|
||||
)
|
||||
def test_korean_polite_is_detected(text):
|
||||
assert detect_politeness(text, "ko") is Politeness.POLITE
|
||||
|
||||
|
||||
@pytest.mark.parametrize("text", ["같이 가자", "빨리 와", "적이 온다", "내가 할게"])
|
||||
def test_korean_casual_is_detected(text):
|
||||
assert detect_politeness(text, "ko") is Politeness.CASUAL
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"text", ["こんにちは、お願いします", "行きますよ", "ありがとうございます", "待ってください"]
|
||||
)
|
||||
def test_japanese_polite_is_detected(text):
|
||||
assert detect_politeness(text, "ja") is Politeness.POLITE
|
||||
|
||||
|
||||
@pytest.mark.parametrize("text", ["早く来いよ", "行くぞ", "危ないだろう"])
|
||||
def test_japanese_casual_is_detected(text):
|
||||
assert detect_politeness(text, "ja") is Politeness.CASUAL
|
||||
|
||||
|
||||
@pytest.mark.parametrize("lang", ["en", "zh"])
|
||||
def test_languages_without_honorifics_are_always_unknown(lang):
|
||||
"""영어·중국어는 문법적 높임이 없으므로 항상 모드를 따라야 한다.
|
||||
|
||||
"please" 를 존댓말 근거로 삼으면 오탐이 너무 많아진다.
|
||||
"""
|
||||
for text in ["Could you please help me, sir?", "Get down now!", "请帮我一下", "快走"]:
|
||||
assert detect_politeness(text, lang) is Politeness.UNKNOWN
|
||||
|
||||
|
||||
def test_empty_text_is_unknown():
|
||||
assert detect_politeness("", "ko") is Politeness.UNKNOWN
|
||||
assert detect_politeness(" ", "ja") is Politeness.UNKNOWN
|
||||
|
||||
|
||||
# --- 반말 변환 --------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("polite", "casual"),
|
||||
[
|
||||
("적이 왼쪽에서 옵니다", "적이 왼쪽에서 와"),
|
||||
("지금 바로 빠지세요", "지금 바로 빠져"),
|
||||
("바론을 치겠습니다", "바론을 칠게"),
|
||||
("적을 눕혔습니다", "적을 눕혔어"),
|
||||
("탄약이 없습니다", "탄약이 없어"),
|
||||
("여기 적이 있습니다", "여기 적이 있어"),
|
||||
("제가 하겠습니다", "제가 할게"),
|
||||
("좋습니다", "좋아"),
|
||||
("빨리 먹습니다", "빨리 먹어"),
|
||||
("저건 함정이에요", "저건 함정이야"),
|
||||
("제 차례예요", "제 차례야"),
|
||||
("조심하세요", "조심해"),
|
||||
("잘 했네요", "잘 했네"),
|
||||
("같이 가죠", "같이 가지"),
|
||||
],
|
||||
)
|
||||
def test_polite_to_casual(polite, casual):
|
||||
assert to_casual(polite) == casual
|
||||
|
||||
|
||||
def test_trailing_yo_is_dropped():
|
||||
assert to_casual("같이 가요") == "같이 가"
|
||||
assert to_casual("빨리 와요!") == "빨리 와!"
|
||||
|
||||
|
||||
def test_mid_sentence_yo_is_preserved():
|
||||
"""'중요', '필요' 의 '요'를 떼면 말이 망가진다."""
|
||||
assert "중요" in to_casual("이게 제일 중요합니다")
|
||||
assert "필요" in to_casual("탄약이 필요합니다")
|
||||
|
||||
|
||||
def test_conversion_is_idempotent_on_already_casual_text():
|
||||
casual = "적이 왼쪽에서 와"
|
||||
assert to_casual(casual) == casual
|
||||
|
||||
|
||||
def test_empty_stays_empty():
|
||||
assert to_casual("") == ""
|
||||
|
||||
|
||||
# --- 모드 적용 규칙 ---------------------------------------------------------
|
||||
|
||||
|
||||
def test_polite_mode_never_alters_output():
|
||||
"""존댓말 모드는 손대지 않는다 — 모델 출력이 이미 격식체다."""
|
||||
text = "적이 왼쪽에서 옵니다"
|
||||
for politeness in Politeness:
|
||||
assert apply_speech_level(text, "ko", SpeechLevel.POLITE, politeness) == text
|
||||
|
||||
|
||||
def test_casual_mode_lowers_when_source_is_unknown():
|
||||
"""영어 원문 → 높임 정보 없음 → 고른 모드(반말)를 따른다."""
|
||||
out = apply_speech_level(
|
||||
"적이 왼쪽에서 옵니다", "ko", SpeechLevel.CASUAL, Politeness.UNKNOWN
|
||||
)
|
||||
assert out == "적이 왼쪽에서 와"
|
||||
|
||||
|
||||
def test_casual_mode_keeps_polite_when_source_was_polite():
|
||||
"""핵심 규칙 — 실제로 존댓말로 말했으면 반말 모드여도 존댓말."""
|
||||
text = "도와주시겠습니까"
|
||||
assert apply_speech_level(text, "ko", SpeechLevel.CASUAL, Politeness.POLITE) == text
|
||||
|
||||
|
||||
def test_casual_mode_lowers_when_source_was_casual():
|
||||
out = apply_speech_level("빨리 옵니다", "ko", SpeechLevel.CASUAL, Politeness.CASUAL)
|
||||
assert out == "빨리 와"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("target", ["en", "ja", "zh"])
|
||||
def test_non_korean_targets_are_untouched(target):
|
||||
"""한국어 외에는 적용할 규칙이 없으므로 원본을 그대로 둔다."""
|
||||
text = "Enemy incoming"
|
||||
assert apply_speech_level(text, target, SpeechLevel.CASUAL, Politeness.UNKNOWN) == text
|
||||
|
||||
|
||||
# --- LLM 프롬프트 지시문 ----------------------------------------------------
|
||||
|
||||
|
||||
def test_prompt_asks_for_polite_when_source_was_polite():
|
||||
hint = prompt_instruction(SpeechLevel.CASUAL, Politeness.POLITE)
|
||||
assert "존댓말" in hint
|
||||
|
||||
|
||||
def test_prompt_asks_for_casual_in_casual_mode():
|
||||
assert "반말" in prompt_instruction(SpeechLevel.CASUAL, Politeness.UNKNOWN)
|
||||
|
||||
|
||||
def test_prompt_asks_for_polite_in_polite_mode():
|
||||
assert "존댓말" in prompt_instruction(SpeechLevel.POLITE, Politeness.UNKNOWN)
|
||||
|
||||
|
||||
# --- 엔진 파이프라인 끝까지 도달하는지 ---------------------------------------
|
||||
|
||||
|
||||
class _PoliteTranslator:
|
||||
"""한국어 격식체를 내놓는 번역기 (NLLB/Seed-X 가 실제로 그렇다)."""
|
||||
|
||||
loaded = True
|
||||
|
||||
def load(self):
|
||||
pass
|
||||
|
||||
def unload(self):
|
||||
pass
|
||||
|
||||
def translate(self, text, source_lang, target_lang, glossary=None, **kwargs):
|
||||
self.kwargs = kwargs
|
||||
return "적이 왼쪽에서 옵니다"
|
||||
|
||||
|
||||
def _run(monkeypatch, tmp_path, level: str, source_text: str, source_lang: str) -> str:
|
||||
from livesub.config import AppConfig
|
||||
from livesub.core.engine import TranslationEngine
|
||||
from livesub.models.asr import Transcript
|
||||
|
||||
cfg = AppConfig()
|
||||
cfg.models.preload_on_start = False
|
||||
cfg.models.speech_level = level
|
||||
cfg.glossary.path = str(tmp_path / "g.json")
|
||||
cfg.glossary.enabled_packs = []
|
||||
|
||||
engine = TranslationEngine(cfg)
|
||||
translator = _PoliteTranslator()
|
||||
|
||||
class Recognizer:
|
||||
def transcribe(self, audio, language=None, fast=False, prompt=""):
|
||||
return Transcript(text=source_text, language=source_lang)
|
||||
|
||||
lines = []
|
||||
engine._on_line = lines.append
|
||||
|
||||
class Segment:
|
||||
is_final = True
|
||||
audio = None
|
||||
ended_at = 0.0
|
||||
|
||||
engine._process(Segment(), Recognizer(), translator, cfg.models)
|
||||
return lines[0].translated_text
|
||||
|
||||
|
||||
def test_english_source_casual_mode_lowers(monkeypatch, tmp_path):
|
||||
"""영어는 높임이 없으니 고른 모드(반말)를 따른다."""
|
||||
assert _run(monkeypatch, tmp_path, "casual", "Enemy from the left", "en") == (
|
||||
"적이 왼쪽에서 와"
|
||||
)
|
||||
|
||||
|
||||
def test_english_source_polite_mode_stays_polite(monkeypatch, tmp_path):
|
||||
assert _run(monkeypatch, tmp_path, "polite", "Enemy from the left", "en") == (
|
||||
"적이 왼쪽에서 옵니다"
|
||||
)
|
||||
|
||||
|
||||
def test_japanese_polite_source_stays_polite_even_in_casual_mode(monkeypatch, tmp_path):
|
||||
"""핵심 규칙 — 실제로 존댓말로 말했으면 반말 모드여도 존댓말."""
|
||||
assert _run(monkeypatch, tmp_path, "casual", "左から来ます", "ja") == (
|
||||
"적이 왼쪽에서 옵니다"
|
||||
)
|
||||
|
||||
|
||||
def test_japanese_casual_source_is_lowered_in_casual_mode(monkeypatch, tmp_path):
|
||||
assert _run(monkeypatch, tmp_path, "casual", "左から来るぞ", "ja") == (
|
||||
"적이 왼쪽에서 와"
|
||||
)
|
||||
|
||||
|
||||
def test_unknown_speech_level_value_falls_back_to_polite(monkeypatch, tmp_path):
|
||||
"""설정 파일이 손상돼도 안전한 존댓말로 떨어져야 한다."""
|
||||
assert _run(monkeypatch, tmp_path, "무엇인가이상한값", "Enemy", "en") == (
|
||||
"적이 왼쪽에서 옵니다"
|
||||
)
|
||||
Reference in New Issue
Block a user